{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/5","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":12,"rows_per_page":100,"rows":[401,500],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding/papers/4","next":"/task/video-understanding/papers/6","papers":[{"url":"/paper/boosting-single-image-super-resolution-via","slug":"boosting-single-image-super-resolution-via","title":"Boosting Single Image Super-Resolution via Partial Channel Shifting","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-referring-relationships-in-videos","slug":"few-shot-referring-relationships-in-videos","title":"Few-Shot Referring Relationships in Videos","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/modeling-video-as-stochastic-processes-for","slug":"modeling-video-as-stochastic-processes-for","title":"Modeling Video As Stochastic Processes for Fine-Grained Video Representation Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-smooth-video-composition","slug":"towards-smooth-video-composition","title":"Towards Smooth Video Composition","date":"2022-12-14","arxiv_id":"2212.07413","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-explainable-video-representation","slug":"contextual-explainable-video-representation","title":"Contextual Explainable Video Representation: Human Perception-based Understanding","date":"2022-12-12","arxiv_id":"2212.06206","repositories_listed":1,"syntology":null},{"url":"/paper/moma-lrg-language-refined-graphs-for-multi","slug":"moma-lrg-language-refined-graphs-for-multi","title":"MOMA-LRG: Language-Refined Graphs for Multi-Object Multi-Actor Activity Parsing","date":"2022-11-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-masked-autoencoders-for-self","slug":"contrastive-masked-autoencoders-for-self","title":"Contrastive Masked Autoencoders for Self-Supervised Video Hashing","date":"2022-11-21","arxiv_id":"2211.11210","repositories_listed":1,"syntology":null},{"url":"/paper/masked-autoencoders-for-egocentric-video","slug":"masked-autoencoders-for-egocentric-video","title":"Masked Autoencoders for Egocentric Video Understanding @ Ego4D Challenge 2022","date":"2022-11-18","arxiv_id":"2211.15286","repositories_listed":1,"syntology":null},{"url":"/paper/vtc-improving-video-text-retrieval-with-user","slug":"vtc-improving-video-text-retrieval-with-user","title":"VTC: Improving Video-Text Retrieval with User Comments","date":"2022-10-19","arxiv_id":"2210.10820","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtc-improving-video-text-retrieval-with-user#ran","syntology_url":"https://syntology.ai/paper/2210.10820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10820"}},"official":null}},{"url":"/paper/how-would-the-viewer-feel-estimating","slug":"how-would-the-viewer-feel-estimating","title":"How Would The Viewer Feel? Estimating Wellbeing From Video Scenarios","date":"2022-10-18","arxiv_id":"2210.10039","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-would-the-viewer-feel-estimating#ran","syntology_url":"https://syntology.ai/paper/2210.10039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10039"}},"official":{"repos":["hendrycks/emodiversity"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egotaskqa-understanding-human-tasks-in","slug":"egotaskqa-understanding-human-tasks-in","title":"EgoTaskQA: Understanding Human Tasks in Egocentric Videos","date":"2022-10-08","arxiv_id":"2210.03929","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egotaskqa-understanding-human-tasks-in#ran","syntology_url":"https://syntology.ai/paper/2210.03929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03929"}},"official":{"repos":["Buzz-Beater/EgoTaskQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-transferable-spatiotemporal","slug":"learning-transferable-spatiotemporal","title":"Learning Transferable Spatiotemporal Representations from Natural Script Knowledge","date":"2022-09-30","arxiv_id":"2209.15280","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-video-temporal-action-segmentation","slug":"streaming-video-temporal-action-segmentation","title":"Streaming Video Temporal Action Segmentation In Real Time","date":"2022-09-28","arxiv_id":"2209.13808","repositories_listed":1,"syntology":null},{"url":"/paper/panoramic-vision-transformer-for-saliency","slug":"panoramic-vision-transformer-for-saliency","title":"Panoramic Vision Transformer for Saliency Detection in 360° Videos","date":"2022-09-19","arxiv_id":"2209.08956","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/panoramic-vision-transformer-for-saliency#ran","syntology_url":"https://syntology.ai/paper/2209.08956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08956"}},"official":{"repos":["hs-yn/paver"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/echocotr-estimation-of-the-left-ventricular","slug":"echocotr-estimation-of-the-left-ventricular","title":"EchoCoTr: Estimation of the Left Ventricular Ejection Fraction from Spatiotemporal Echocardiography","date":"2022-09-09","arxiv_id":"2209.04242","repositories_listed":1,"syntology":null},{"url":"/paper/point-primitive-transformer-for-long-term-4d","slug":"point-primitive-transformer-for-long-term-4d","title":"Point Primitive Transformer for Long-Term 4D Point Cloud Video Understanding","date":"2022-07-30","arxiv_id":"2208.00281","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/point-primitive-transformer-for-long-term-4d#ran","syntology_url":"https://syntology.ai/paper/2208.00281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.00281"}},"official":{"repos":["hoi4d/PPTr"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/static-and-dynamic-concepts-for-self","slug":"static-and-dynamic-concepts-for-self","title":"Static and Dynamic Concepts for Self-supervised Video Representation Learning","date":"2022-07-26","arxiv_id":"2207.12795","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/static-and-dynamic-concepts-for-self#ran","syntology_url":"https://syntology.ai/paper/2207.12795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12795"}},"official":{"repos":["shvdiwnkozbw/Self-supervised-Video-Concept"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/clover-towards-a-unified-video-language","slug":"clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","arxiv_id":"2207.07885","repositories_listed":1,"syntology":null},{"url":"/paper/is-appearance-free-action-recognition","slug":"is-appearance-free-action-recognition","title":"Is Appearance Free Action Recognition Possible?","date":"2022-07-13","arxiv_id":"2207.06261","repositories_listed":1,"syntology":null},{"url":"/paper/un-likelihood-training-for-interpretable","slug":"un-likelihood-training-for-interpretable","title":"(Un)likelihood Training for Interpretable Embedding","date":"2022-07-01","arxiv_id":"2207.00282","repositories_listed":1,"syntology":null},{"url":"/paper/technical-report-for-cvpr-2022-loveu-aqtc","slug":"technical-report-for-cvpr-2022-loveu-aqtc","title":"Technical Report for CVPR 2022 LOVEU AQTC Challenge","date":"2022-06-29","arxiv_id":"2206.14555","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-efficient-image-to-video-transfer","slug":"parameter-efficient-image-to-video-transfer","title":"ST-Adapter: Parameter-Efficient Image-to-Video Transfer Learning","date":"2022-06-27","arxiv_id":"2206.13559","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/parameter-efficient-image-to-video-transfer#ran","syntology_url":"https://syntology.ai/paper/2206.13559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13559"}},"official":{"repos":["linziyi96/st-adapter"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reveca-rich-encoder-decoder-framework-for","slug":"reveca-rich-encoder-decoder-framework-for","title":"REVECA -- Rich Encoder-decoder framework for Video Event CAptioner","date":"2022-06-18","arxiv_id":"2206.09178","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-dialogue-state-tracking-1","slug":"multimodal-dialogue-state-tracking-1","title":"Multimodal Dialogue State Tracking","date":"2022-06-16","arxiv_id":"2206.07898","repositories_listed":1,"syntology":null},{"url":"/paper/stand-alone-inter-frame-attention-in-video-1","slug":"stand-alone-inter-frame-attention-in-video-1","title":"Stand-Alone Inter-Frame Attention in Video Models","date":"2022-06-14","arxiv_id":"2206.06931","repositories_listed":1,"syntology":null},{"url":"/paper/minimum-efforts-to-build-an-end-to-end","slug":"minimum-efforts-to-build-an-end-to-end","title":"A Simple and Efficient Pipeline to Build an End-to-End Spatial-Temporal Action Detector","date":"2022-06-07","arxiv_id":"2206.03064","repositories_listed":1,"syntology":null},{"url":"/paper/from-representation-to-reasoning-towards-both","slug":"from-representation-to-reasoning-towards-both","title":"From Representation to Reasoning: Towards both Evidence and Commonsense Reasoning for Video Question-Answering","date":"2022-05-30","arxiv_id":"2205.14895","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-representation-to-reasoning-towards-both#ran","syntology_url":"https://syntology.ai/paper/2205.14895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14895"}},"official":{"repos":["bcmi/causal-vidqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/free-lunch-for-surgical-video-understanding","slug":"free-lunch-for-surgical-video-understanding","title":"Free Lunch for Surgical Video Understanding by Distilling Self-Supervisions","date":"2022-05-19","arxiv_id":"2205.09292","repositories_listed":1,"syntology":null},{"url":"/paper/etad-a-unified-framework-for-efficient","slug":"etad-a-unified-framework-for-efficient","title":"ETAD: Training Action Detection End to End on a Laptop","date":"2022-05-14","arxiv_id":"2205.07134","repositories_listed":1,"syntology":null},{"url":"/paper/a-multi-person-video-dataset-annotation","slug":"a-multi-person-video-dataset-annotation","title":"A Multi-Person Video Dataset Annotation Method of Spatio-Temporally Actions","date":"2022-04-21","arxiv_id":"2204.10160","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-end-to-end-temporal","slug":"an-empirical-study-of-end-to-end-temporal","title":"An Empirical Study of End-to-End Temporal Action Detection","date":"2022-04-06","arxiv_id":"2204.02932","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-empirical-study-of-end-to-end-temporal#ran","syntology_url":"https://syntology.ai/paper/2204.02932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02932"}},"official":{"repos":["xlliu7/E2E-TAD"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-alignment-networks-for-long-term","slug":"temporal-alignment-networks-for-long-term","title":"Temporal Alignment Networks for Long-term Video","date":"2022-04-06","arxiv_id":"2204.02968","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/temporal-alignment-networks-for-long-term#ran","syntology_url":"https://syntology.ai/paper/2204.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02968"}},"official":null}},{"url":"/paper/long-movie-clip-classification-with-state","slug":"long-movie-clip-classification-with-state","title":"Long Movie Clip Classification with State-Space Video Models","date":"2022-04-04","arxiv_id":"2204.01692","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":2,"n_ran_checked":3,"n_instrument":8,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/long-movie-clip-classification-with-state#ran","syntology_url":"https://syntology.ai/paper/2204.01692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01692"}},"official":{"repos":["md-mohaiminul/ViS4mer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pyskl-a-toolbox-for-skeleton-based-video","slug":"pyskl-a-toolbox-for-skeleton-based-video","title":"PYSKL: a toolbox for skeleton-based video understanding","date":"2022-04-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/spact-self-supervised-privacy-preservation","slug":"spact-self-supervised-privacy-preservation","title":"SPAct: Self-supervised Privacy Preservation for Action Recognition","date":"2022-03-29","arxiv_id":"2203.15205","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/spact-self-supervised-privacy-preservation#ran","syntology_url":"https://syntology.ai/paper/2203.15205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15205"}},"official":{"repos":["daveishan/spact"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-severe-is-benchmark-sensitivity-in-video","slug":"how-severe-is-benchmark-sensitivity-in-video","title":"How Severe is Benchmark-Sensitivity in Video Self-Supervised Learning?","date":"2022-03-27","arxiv_id":"2203.14221","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-pitfalls-of-batch-normalization-for","slug":"on-the-pitfalls-of-batch-normalization-for","title":"On the Pitfalls of Batch Normalization for End-to-End Video Learning: A Study on Surgical Workflow Analysis","date":"2022-03-15","arxiv_id":"2203.07976","repositories_listed":1,"syntology":null},{"url":"/paper/domain-knowledge-informed-self-supervised","slug":"domain-knowledge-informed-self-supervised","title":"Domain Knowledge-Informed Self-Supervised Representations for Workout Form Assessment","date":"2022-02-28","arxiv_id":"2202.14019","repositories_listed":1,"syntology":null},{"url":"/paper/actionformer-localizing-moments-of-actions","slug":"actionformer-localizing-moments-of-actions","title":"ActionFormer: Localizing Moments of Actions with Transformers","date":"2022-02-16","arxiv_id":"2202.07925","repositories_listed":1,"syntology":null},{"url":"/paper/learning-optical-flow-with-adaptive-graph","slug":"learning-optical-flow-with-adaptive-graph","title":"Learning Optical Flow with Adaptive Graph Reasoning","date":"2022-02-08","arxiv_id":"2202.03857","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-optical-flow-with-adaptive-graph#ran","syntology_url":"https://syntology.ai/paper/2202.03857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03857"}},"official":{"repos":["la30/agflow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/perceptual-coding-for-compressed-video","slug":"perceptual-coding-for-compressed-video","title":"A Coding Framework and Benchmark towards Low-Bitrate Video Understanding","date":"2022-02-06","arxiv_id":"2202.02813","repositories_listed":1,"syntology":null},{"url":"/paper/capturing-temporal-information-in-a-single","slug":"capturing-temporal-information-in-a-single","title":"Capturing Temporal Information in a Single Frame: Channel Sampling Strategies for Action Recognition","date":"2022-01-25","arxiv_id":"2201.10394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/capturing-temporal-information-in-a-single#ran","syntology_url":"https://syntology.ai/paper/2201.10394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.10394"}},"official":{"repos":["kiyoon/channel_sampling"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiview-transformers-for-video-recognition","slug":"multiview-transformers-for-video-recognition","title":"Multiview Transformers for Video Recognition","date":"2022-01-12","arxiv_id":"2201.04288","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-long-term-dependencies-for","slug":"exploiting-long-term-dependencies-for","title":"Exploiting Long-Term Dependencies for Generating Dynamic Scene Graphs","date":"2021-12-18","arxiv_id":"2112.09828","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-spatio-temporal-pretext-learning","slug":"contrastive-spatio-temporal-pretext-learning","title":"Contrastive Spatio-Temporal Pretext Learning for Self-supervised Video Representation","date":"2021-12-16","arxiv_id":"2112.08913","repositories_listed":1,"syntology":null},{"url":"/paper/prompting-visual-language-models-for","slug":"prompting-visual-language-models-for","title":"Prompting Visual-Language Models for Efficient Video Understanding","date":"2021-12-08","arxiv_id":"2112.04478","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prompting-visual-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2112.04478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.04478"}},"official":{"repos":["ju-chen/Efficient-Prompt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/suppressing-static-visual-cues-via","slug":"suppressing-static-visual-cues-via","title":"Suppressing Static Visual Cues via Normalizing Flows for Self-Supervised Video Representation Learning","date":"2021-12-07","arxiv_id":"2112.03803","repositories_listed":1,"syntology":null},{"url":"/paper/tokenlearner-adaptive-space-time-tokenization","slug":"tokenlearner-adaptive-space-time-tokenization","title":"TokenLearner: Adaptive Space-Time Tokenization for Videos","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/swinbert-end-to-end-transformers-with-sparse","slug":"swinbert-end-to-end-transformers-with-sparse","title":"SwinBERT: End-to-End Transformers with Sparse Attention for Video Captioning","date":"2021-11-25","arxiv_id":"2111.13196","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swinbert-end-to-end-transformers-with-sparse#ran","syntology_url":"https://syntology.ai/paper/2111.13196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13196"}},"official":{"repos":["microsoft/swinbert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-pyramid-multimodal-pyramid-attentional","slug":"mm-pyramid-multimodal-pyramid-attentional","title":"MM-Pyramid: Multimodal Pyramid Attentional Network for Audio-Visual Event Localization and Video Parsing","date":"2021-11-24","arxiv_id":"2111.12374","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-pyramid-multimodal-pyramid-attentional#ran","syntology_url":"https://syntology.ai/paper/2111.12374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12374"}},"official":{"repos":["JustinYuu/MM_Pyramid"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/violet-end-to-end-video-language-transformers","slug":"violet-end-to-end-video-language-transformers","title":"VIOLET : End-to-End Video-Language Transformers with Masked Visual-token Modeling","date":"2021-11-24","arxiv_id":"2111.12681","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/violet-end-to-end-video-language-transformers#ran","syntology_url":"https://syntology.ai/paper/2111.12681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12681"}},"official":{"repos":["tsujuifu/pytorch_violet"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/pytorchvideo-a-deep-learning-library-for","slug":"pytorchvideo-a-deep-learning-library-for","title":"PyTorchVideo: A Deep Learning Library for Video Understanding","date":"2021-11-18","arxiv_id":"2111.09887","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pytorchvideo-a-deep-learning-library-for#ran","syntology_url":"https://syntology.ai/paper/2111.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.09887"}},"official":{"repos":["facebookresearch/pytorchvideo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-mechanisms-in-computer-vision-a","slug":"attention-mechanisms-in-computer-vision-a","title":"Attention Mechanisms in Computer Vision: A Survey","date":"2021-11-15","arxiv_id":"2111.07624","repositories_listed":1,"syntology":null},{"url":"/paper/relational-self-attention-what-s-missing-in","slug":"relational-self-attention-what-s-missing-in","title":"Relational Self-Attention: What's Missing in Attention for Video Understanding","date":"2021-11-02","arxiv_id":"2111.01673","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/relational-self-attention-what-s-missing-in#ran","syntology_url":"https://syntology.ai/paper/2111.01673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.01673"}},"official":{"repos":["KimManjin/RSA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-spatio-temporal-layouts-for","slug":"revisiting-spatio-temporal-layouts-for","title":"Revisiting spatio-temporal layouts for compositional action recognition","date":"2021-11-02","arxiv_id":"2111.01936","repositories_listed":1,"syntology":null},{"url":"/paper/re-id-ar-improved-person-re-identification-in","slug":"re-id-ar-improved-person-re-identification-in","title":"Re-ID-AR: Improved Person Re-identification in Video via Joint Weakly Supervised Action Recognition","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-the-robustness-of-spatial","slug":"benchmarking-the-robustness-of-spatial","title":"Benchmarking the Robustness of Spatial-Temporal Models Against Corruptions","date":"2021-10-13","arxiv_id":"2110.06513","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/benchmarking-the-robustness-of-spatial#ran","syntology_url":"https://syntology.ai/paper/2110.06513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06513"}},"official":{"repos":["newbeeyoung/video-corruption-robustness"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/noisyactions2m-a-multimedia-dataset-for-video","slug":"noisyactions2m-a-multimedia-dataset-for-video","title":"NoisyActions2M: A Multimedia Dataset for Video Understanding from Noisy Labels","date":"2021-10-13","arxiv_id":"2110.06827","repositories_listed":1,"syntology":null},{"url":"/paper/object-region-video-transformers-1","slug":"object-region-video-transformers-1","title":"Object-Region Video Transformers","date":"2021-10-13","arxiv_id":"2110.06915","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/object-region-video-transformers-1#ran","syntology_url":"https://syntology.ai/paper/2110.06915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06915"}},"official":null}},{"url":"/paper/intentvizor-towards-generic-query-guided","slug":"intentvizor-towards-generic-query-guided","title":"IntentVizor: Towards Generic Query Guided Interactive Video Summarization","date":"2021-09-30","arxiv_id":"2109.14834","repositories_listed":1,"syntology":null},{"url":"/paper/pairwise-emotional-relationship-recognition","slug":"pairwise-emotional-relationship-recognition","title":"Pairwise Emotional Relationship Recognition in Drama Videos: Dataset and Benchmark","date":"2021-09-23","arxiv_id":"2109.11243","repositories_listed":1,"syntology":null},{"url":"/paper/towards-high-quality-temporal-action","slug":"towards-high-quality-temporal-action","title":"Towards High-Quality Temporal Action Detection with Sparse Proposals","date":"2021-09-18","arxiv_id":"2109.08847","repositories_listed":1,"syntology":null},{"url":"/paper/spatio-temporal-perturbations-for-video","slug":"spatio-temporal-perturbations-for-video","title":"Spatio-Temporal Perturbations for Video Attribution","date":"2021-09-01","arxiv_id":"2109.00222","repositories_listed":1,"syntology":null},{"url":"/paper/ligar-lightweight-general-purpose-action","slug":"ligar-lightweight-general-purpose-action","title":"LIGAR: Lightweight General-purpose Action Recognition","date":"2021-08-30","arxiv_id":"2108.13153","repositories_listed":1,"syntology":null},{"url":"/paper/foreground-action-consistency-network-for","slug":"foreground-action-consistency-network-for","title":"Foreground-Action Consistency Network for Weakly Supervised Temporal Action Localization","date":"2021-08-14","arxiv_id":"2108.06524","repositories_listed":1,"syntology":null},{"url":"/paper/autovideo-an-automated-video-action","slug":"autovideo-an-automated-video-action","title":"AutoVideo: An Automated Video Action Recognition System","date":"2021-08-09","arxiv_id":"2108.04212","repositories_listed":1,"syntology":null},{"url":"/paper/elaborative-rehearsal-for-zero-shot-action","slug":"elaborative-rehearsal-for-zero-shot-action","title":"Elaborative Rehearsal for Zero-shot Action Recognition","date":"2021-08-05","arxiv_id":"2108.02833","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/elaborative-rehearsal-for-zero-shot-action#ran","syntology_url":"https://syntology.ai/paper/2108.02833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02833"}},"official":{"repos":["DeLightCMU/ElaborativeRehearsal"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-self-supervised-video","slug":"enhancing-self-supervised-video","title":"Enhancing Self-supervised Video Representation Learning via Multi-level Feature Optimization","date":"2021-08-04","arxiv_id":"2108.02183","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-self-supervised-video#ran","syntology_url":"https://syntology.ai/paper/2108.02183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02183"}},"official":{"repos":["shvdiwnkozbw/video-representation-via-multi-level-optimization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/an-image-classifier-can-suffice-video","slug":"an-image-classifier-can-suffice-video","title":"Can An Image Classifier Suffice For Action Recognition?","date":"2021-06-26","arxiv_id":"2106.14104","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-image-classifier-can-suffice-video#ran","syntology_url":"https://syntology.ai/paper/2106.14104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14104"}},"official":{"repos":["ibm/sifar-pytorch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vimpac-video-pre-training-via-masked-token","slug":"vimpac-video-pre-training-via-masked-token","title":"VIMPAC: Video Pre-Training via Masked Token Prediction and Contrastive Learning","date":"2021-06-21","arxiv_id":"2106.11250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vimpac-video-pre-training-via-masked-token#ran","syntology_url":"https://syntology.ai/paper/2106.11250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11250"}},"official":{"repos":["airsplay/vimpac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-the-predictability-of-the-future","slug":"learning-the-predictability-of-the-future","title":"Learning the Predictability of the Future","date":"2021-06-19","arxiv_id":"2101.01600","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/learning-the-predictability-of-the-future#ran","syntology_url":"https://syntology.ai/paper/2101.01600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.01600"}},"official":{"repos":["cvlab-columbia/hyperfuture"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/next-qa-next-phase-of-question-answering-to-1","slug":"next-qa-next-phase-of-question-answering-to-1","title":"NExT-QA: Next Phase of Question-Answering to Explaining Temporal Actions","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-temporal-action-detection-with","slug":"end-to-end-temporal-action-detection-with","title":"End-to-end Temporal Action Detection with Transformer","date":"2021-06-18","arxiv_id":"2106.10271","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-temporal-action-detection-with#ran","syntology_url":"https://syntology.ai/paper/2106.10271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.10271"}},"official":{"repos":["xlliu7/TadTR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/isolated-sign-recognition-from-rgb-video","slug":"isolated-sign-recognition-from-rgb-video","title":"Isolated Sign Recognition from RGB Video using Pose Flow and Self-Attention","date":"2021-06-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vt-ssum-a-benchmark-dataset-for-video","slug":"vt-ssum-a-benchmark-dataset-for-video","title":"VT-SSum: A Benchmark Dataset for Video Transcript Segmentation and Summarization","date":"2021-06-10","arxiv_id":"2106.05606","repositories_listed":1,"syntology":null},{"url":"/paper/towards-training-stronger-video-vision","slug":"towards-training-stronger-video-vision","title":"Towards Training Stronger Video Vision Transformers for EPIC-KITCHENS-100 Action Recognition","date":"2021-06-09","arxiv_id":"2106.05058","repositories_listed":1,"syntology":null},{"url":"/paper/technical-report-temporal-aggregate","slug":"technical-report-temporal-aggregate","title":"Technical Report: Temporal Aggregate Representations","date":"2021-06-06","arxiv_id":"2106.03152","repositories_listed":1,"syntology":null},{"url":"/paper/fineaction-a-fined-video-dataset-for-temporal","slug":"fineaction-a-fined-video-dataset-for-temporal","title":"FineAction: A Fine-Grained Video Dataset for Temporal Action Localization","date":"2021-05-24","arxiv_id":"2105.11107","repositories_listed":1,"syntology":null},{"url":"/paper/vlm-task-agnostic-video-language-model-pre","slug":"vlm-task-agnostic-video-language-model-pre","title":"VLM: Task-agnostic Video-Language Model Pre-training for Video Understanding","date":"2021-05-20","arxiv_id":"2105.09996","repositories_listed":1,"syntology":null},{"url":"/paper/multisports-a-multi-person-video-dataset-of","slug":"multisports-a-multi-person-video-dataset-of","title":"MultiSports: A Multi-Person Video Dataset of Spatio-Temporally Localized Sports Actions","date":"2021-05-16","arxiv_id":"2105.07404","repositories_listed":1,"syntology":null},{"url":"/paper/relation-aware-hierarchical-attention","slug":"relation-aware-hierarchical-attention","title":"Relation-aware Hierarchical Attention Framework for Video Question Answering","date":"2021-05-13","arxiv_id":"2105.06160","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-image-to-video-synthesis-using","slug":"stochastic-image-to-video-synthesis-using","title":"Stochastic Image-to-Video Synthesis using cINNs","date":"2021-05-10","arxiv_id":"2105.04551","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":4,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stochastic-image-to-video-synthesis-using#ran","syntology_url":"https://syntology.ai/paper/2105.04551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.04551"}},"official":{"repos":["CompVis/image2video-synthesis-using-cINNs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/frameexit-conditional-early-exiting-for","slug":"frameexit-conditional-early-exiting-for","title":"FrameExit: Conditional Early Exiting for Efficient Video Recognition","date":"2021-04-27","arxiv_id":"2104.13400","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":1,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frameexit-conditional-early-exiting-for#ran","syntology_url":"https://syntology.ai/paper/2104.13400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.13400"}},"official":{"repos":["Qualcomm-AI-research/FrameExit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-smooth-online-action-detection","slug":"temporally-smooth-online-action-detection","title":"Temporally smooth online action detection using cycle-consistent future anticipation","date":"2021-04-16","arxiv_id":"2104.08030","repositories_listed":1,"syntology":null},{"url":"/paper/crossover-learning-for-fast-online-video","slug":"crossover-learning-for-fast-online-video","title":"Crossover Learning for Fast Online Video Instance Segmentation","date":"2021-04-13","arxiv_id":"2104.05970","repositories_listed":1,"syntology":null},{"url":"/paper/fill-in-the-blank-as-a-challenging-video","slug":"fill-in-the-blank-as-a-challenging-video","title":"FIBER: Fill-in-the-Blanks as a Challenging Video Understanding Evaluation Framework","date":"2021-04-09","arxiv_id":"2104.04182","repositories_listed":1,"syntology":null},{"url":"/paper/tuber-tube-transformer-for-action-detection","slug":"tuber-tube-transformer-for-action-detection","title":"TubeR: Tubelet Transformer for Video Action Detection","date":"2021-04-02","arxiv_id":"2104.00969","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tuber-tube-transformer-for-action-detection#ran","syntology_url":"https://syntology.ai/paper/2104.00969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00969"}},"official":null}},{"url":"/paper/visual-semantic-role-labeling-for-video","slug":"visual-semantic-role-labeling-for-video","title":"Visual Semantic Role Labeling for Video Understanding","date":"2021-04-02","arxiv_id":"2104.00990","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-semantic-role-labeling-for-video#ran","syntology_url":"https://syntology.ai/paper/2104.00990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00990"}},"official":{"repos":["TheShadow29/VidSitu"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-salient-boundary-feature-for-anchor","slug":"learning-salient-boundary-feature-for-anchor","title":"Learning Salient Boundary Feature for Anchor-free Temporal Action Localization","date":"2021-03-24","arxiv_id":"2103.13137","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-context-aggregation-network-for","slug":"temporal-context-aggregation-network-for","title":"Temporal Context Aggregation Network for Temporal Action Proposal Refinement","date":"2021-03-24","arxiv_id":"2103.13141","repositories_listed":1,"syntology":null},{"url":"/paper/temporally-weighted-hierarchical-clustering","slug":"temporally-weighted-hierarchical-clustering","title":"Temporally-Weighted Hierarchical Clustering for Unsupervised Action Segmentation","date":"2021-03-20","arxiv_id":"2103.11264","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/temporally-weighted-hierarchical-clustering#ran","syntology_url":"https://syntology.ai/paper/2103.11264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.11264"}},"official":{"repos":["ssarfraz/FINCH-CLustering"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/win-fail-action-recognition","slug":"win-fail-action-recognition","title":"Win-Fail Action Recognition","date":"2021-02-15","arxiv_id":"2102.07355","repositories_listed":1,"syntology":null},{"url":"/paper/learning-self-similarity-in-space-and-time-as-1","slug":"learning-self-similarity-in-space-and-time-as-1","title":"Learning Self-Similarity in Space and Time as Generalized Motion for Video Action Recognition","date":"2021-02-14","arxiv_id":"2102.07092","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-self-similarity-in-space-and-time-as-1#ran","syntology_url":"https://syntology.ai/paper/2102.07092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07092"}},"official":{"repos":["arunos728/SELFY"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tclr-temporal-contrastive-learning-for-video","slug":"tclr-temporal-contrastive-learning-for-video","title":"TCLR: Temporal Contrastive Learning for Video Representation","date":"2021-01-20","arxiv_id":"2101.07974","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/tclr-temporal-contrastive-learning-for-video#ran","syntology_url":"https://syntology.ai/paper/2101.07974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.07974"}},"official":{"repos":["DAVEISHAN/TCLR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learning-self-similarity-in-space-and-time-as","slug":"learning-self-similarity-in-space-and-time-as","title":"Learning Self-Similarity in Space and Time as a Generalized Motion for Action Recognition","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-study-of-deep-video-action","slug":"a-comprehensive-study-of-deep-video-action","title":"A Comprehensive Study of Deep Video Action Recognition","date":"2020-12-11","arxiv_id":"2012.06567","repositories_listed":1,"syntology":null},{"url":"/paper/video-understanding-based-on-human-action-and","slug":"video-understanding-based-on-human-action-and","title":"Improved Actor Relation Graph based Group Activity Recognition","date":"2020-10-24","arxiv_id":"2010.12968","repositories_listed":1,"syntology":null},{"url":"/paper/video-action-understanding-a-tutorial","slug":"video-action-understanding-a-tutorial","title":"Video Action Understanding","date":"2020-10-13","arxiv_id":"2010.06647","repositories_listed":1,"syntology":null},{"url":"/paper/the-end-of-end-to-end-a-video-understanding","slug":"the-end-of-end-to-end-a-video-understanding","title":"The End-of-End-to-End: A Video Understanding Pentathlon Challenge (2020)","date":"2020-08-03","arxiv_id":"2008.00744","repositories_listed":1,"syntology":null},{"url":"/paper/video-moment-localization-using-object","slug":"video-moment-localization-using-object","title":"Video Moment Localization using Object Evidence and Reverse Captioning","date":"2020-06-18","arxiv_id":"2006.10260","repositories_listed":1,"syntology":null}],"record_sha256":"7d36730ed79c08714cb16181f18d0a84dc21775c6a05c6aafcea482b362b63b0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}