{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/ran/2","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":218,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding/papers/ran/1","prev":"/task/video-understanding/papers/ran/1","next":"/task/video-understanding/papers/ran/3","papers":[{"url":"/paper/videotree-adaptive-tree-based-video","slug":"videotree-adaptive-tree-based-video","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","date":"2024-05-29","arxiv_id":"2405.19209","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videotree-adaptive-tree-based-video#ran","syntology_url":"https://syntology.ai/paper/2405.19209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19209"}},"official":{"repos":["Ziyang412/VideoTree"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hawk-learning-to-understand-open-world-video","slug":"hawk-learning-to-understand-open-world-video","title":"Hawk: Learning to Understand Open-World Video Anomalies","date":"2024-05-27","arxiv_id":"2405.16886","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hawk-learning-to-understand-open-world-video#ran","syntology_url":"https://syntology.ai/paper/2405.16886","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16886"}},"official":{"repos":["jqtangust/hawk"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dense-connector-for-mllms","slug":"dense-connector-for-mllms","title":"Dense Connector for MLLMs","date":"2024-05-22","arxiv_id":"2405.13800","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dense-connector-for-mllms#ran","syntology_url":"https://syntology.ai/paper/2405.13800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.13800"}},"official":{"repos":["HJYao00/DenseConnector"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/topa-extend-large-language-models-for-video","slug":"topa-extend-large-language-models-for-video","title":"TOPA: Extending Large Language Models for Video Understanding via Text-Only Pre-Alignment","date":"2024-05-22","arxiv_id":"2405.13911","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":3,"n_ran_checked":4,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":3,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/topa-extend-large-language-models-for-video#ran","syntology_url":"https://syntology.ai/paper/2405.13911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.13911"}},"official":{"repos":["dhg-wei/topa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/moviechat-question-aware-sparse-memory-for","slug":"moviechat-question-aware-sparse-memory-for","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","date":"2024-04-26","arxiv_id":"2404.17176","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/moviechat-question-aware-sparse-memory-for#ran","syntology_url":"https://syntology.ai/paper/2404.17176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17176"}},"official":{"repos":["rese1f/MovieChat"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-temporal-contextualization-for","slug":"leveraging-temporal-contextualization-for","title":"Leveraging Temporal Contextualization for Video Action Recognition","date":"2024-04-15","arxiv_id":"2404.09490","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/leveraging-temporal-contextualization-for#ran","syntology_url":"https://syntology.ai/paper/2404.09490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09490"}},"official":{"repos":["naver-ai/tc-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/task-driven-exploration-decoupling-and-inter","slug":"task-driven-exploration-decoupling-and-inter","title":"Task-Driven Exploration: Decoupling and Inter-Task Feedback for Joint Moment Retrieval and Highlight Detection","date":"2024-04-14","arxiv_id":"2404.09263","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":5,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/task-driven-exploration-decoupling-and-inter#ran","syntology_url":"https://syntology.ai/paper/2404.09263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09263"}},"official":{"repos":["edengabriel/taskweave"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/in-my-perspective-in-my-hands-accurate","slug":"in-my-perspective-in-my-hands-accurate","title":"In My Perspective, In My Hands: Accurate Egocentric 2D Hand Pose and Action Recognition","date":"2024-04-14","arxiv_id":"2404.09308","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/in-my-perspective-in-my-hands-accurate#ran","syntology_url":"https://syntology.ai/paper/2404.09308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09308"}},"official":{"repos":["wiktormucha/effhandegonet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt4-video-advancing-multimodal-llms-for","slug":"minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","arxiv_id":"2404.03413","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt4-video-advancing-multimodal-llms-for#ran","syntology_url":"https://syntology.ai/paper/2404.03413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03413"}},"official":{"repos":["Vision-CAIR/MiniGPT4-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/snag-scalable-and-accurate-video-grounding","slug":"snag-scalable-and-accurate-video-grounding","title":"SnAG: Scalable and Accurate Video Grounding","date":"2024-04-02","arxiv_id":"2404.02257","repositories_listed":1,"syntology":{"n":20,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/snag-scalable-and-accurate-video-grounding#ran","syntology_url":"https://syntology.ai/paper/2404.02257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02257"}},"official":null}},{"url":"/paper/st-llm-large-language-models-are-effective-1","slug":"st-llm-large-language-models-are-effective-1","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","date":"2024-03-30","arxiv_id":"2404.00308","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/st-llm-large-language-models-are-effective-1#ran","syntology_url":"https://syntology.ai/paper/2404.00308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00308"}},"official":{"repos":["TencentARC/ST-LLM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/omnivid-a-generative-framework-for-universal#ran","syntology_url":"https://syntology.ai/paper/2403.17935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17935"}},"official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-long-videos-in-one-multimodal","slug":"understanding-long-videos-in-one-multimodal","title":"Understanding Long Videos with Multimodal Language Models","date":"2024-03-25","arxiv_id":"2403.16998","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-long-videos-in-one-multimodal#ran","syntology_url":"https://syntology.ai/paper/2403.16998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16998"}},"official":{"repos":["kahnchana/mvu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/language-repository-for-long-video","slug":"language-repository-for-long-video","title":"Language Repository for Long Video Understanding","date":"2024-03-21","arxiv_id":"2403.14622","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-repository-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2403.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14622"}},"official":{"repos":["kkahatapitiya/langrepo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-pre-trained-text-to-video-diffusion","slug":"exploring-pre-trained-text-to-video-diffusion","title":"Exploring Pre-trained Text-to-Video Diffusion Models for Referring Video Object Segmentation","date":"2024-03-18","arxiv_id":"2403.12042","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":5,"n_ran_checked":10,"n_instrument":6,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":20,"phrase":"16 ran (of which 5 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/exploring-pre-trained-text-to-video-diffusion#ran","syntology_url":"https://syntology.ai/paper/2403.12042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12042"}},"official":{"repos":["buxiangzhiren/vd-it"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":5,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/video-mamba-suite-state-space-model-as-a","slug":"video-mamba-suite-state-space-model-as-a","title":"Video Mamba Suite: State Space Model as a Versatile Alternative for Video Understanding","date":"2024-03-14","arxiv_id":"2403.09626","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/video-mamba-suite-state-space-model-as-a#ran","syntology_url":"https://syntology.ai/paper/2403.09626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09626"}},"official":{"repos":["opengvlab/video-mamba-suite"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-image-is-worth-1-2-tokens-after-layer-2","slug":"an-image-is-worth-1-2-tokens-after-layer-2","title":"An Image is Worth 1/2 Tokens After Layer 2: Plug-and-Play Inference Acceleration for Large Vision-Language Models","date":"2024-03-11","arxiv_id":"2403.06764","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-image-is-worth-1-2-tokens-after-layer-2#ran","syntology_url":"https://syntology.ai/paper/2403.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06764"}},"official":{"repos":["pkunlp-icler/fastv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/videomamba-state-space-model-for-efficient","slug":"videomamba-state-space-model-for-efficient","title":"VideoMamba: State Space Model for Efficient Video Understanding","date":"2024-03-11","arxiv_id":"2403.06977","repositories_listed":2,"syntology":{"n":15,"n_ran":9,"n_constructed":5,"n_ran_checked":6,"n_instrument":3,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/videomamba-state-space-model-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2403.06977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06977"}},"official":{"repos":["opengvlab/videomamba"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/beyond-mot-semantic-multi-object-tracking","slug":"beyond-mot-semantic-multi-object-tracking","title":"Beyond MOT: Semantic Multi-Object Tracking","date":"2024-03-08","arxiv_id":"2403.05021","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/beyond-mot-semantic-multi-object-tracking#ran","syntology_url":"https://syntology.ai/paper/2403.05021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05021"}},"official":{"repos":["Nathan-Li123/SMOTer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-recap-recursive-captioning-of-hour-long","slug":"video-recap-recursive-captioning-of-hour-long","title":"Video ReCap: Recursive Captioning of Hour-Long Videos","date":"2024-02-20","arxiv_id":"2402.13250","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-recap-recursive-captioning-of-hour-long#ran","syntology_url":"https://syntology.ai/paper/2402.13250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13250"}},"official":{"repos":["md-mohaiminul/VideoRecap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/world-model-on-million-length-video-and","slug":"world-model-on-million-length-video-and","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","date":"2024-02-13","arxiv_id":"2402.08268","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-model-on-million-length-video-and#ran","syntology_url":"https://syntology.ai/paper/2402.08268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08268"}},"official":{"repos":["LargeWorldModel/LWM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-granularity-correspondence-learning-1","slug":"multi-granularity-correspondence-learning-1","title":"Multi-granularity Correspondence Learning from Long-term Noisy Videos","date":"2024-01-30","arxiv_id":"2401.16702","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-granularity-correspondence-learning-1#ran","syntology_url":"https://syntology.ai/paper/2401.16702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.16702"}},"official":null}},{"url":"/paper/a-simple-llm-framework-for-long-range-video","slug":"a-simple-llm-framework-for-long-range-video","title":"A Simple LLM Framework for Long-Range Video Question-Answering","date":"2023-12-28","arxiv_id":"2312.17235","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-llm-framework-for-long-range-video#ran","syntology_url":"https://syntology.ai/paper/2312.17235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17235"}},"official":{"repos":["ceezh/llovi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shot2story20k-a-new-benchmark-for","slug":"shot2story20k-a-new-benchmark-for","title":"Shot2Story20K: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","date":"2023-12-16","arxiv_id":"2312.10300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shot2story20k-a-new-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2312.10300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10300"}},"official":{"repos":["bytedance/Shot2Story"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/smile-multimodal-dataset-for-understanding","slug":"smile-multimodal-dataset-for-understanding","title":"SMILE: Multimodal Dataset for Understanding Laughter in Video with Language Models","date":"2023-12-15","arxiv_id":"2312.09818","repositories_listed":2,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/smile-multimodal-dataset-for-understanding#ran","syntology_url":"https://syntology.ai/paper/2312.09818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09818"}},"official":{"repos":["postech-ami/smile-dataset","smile-data/smile"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/x4d-sceneformer-enhanced-scene-understanding","slug":"x4d-sceneformer-enhanced-scene-understanding","title":"X4D-SceneFormer: Enhanced Scene Understanding on 4D Point Cloud Videos through Cross-modal Knowledge Transfer","date":"2023-12-12","arxiv_id":"2312.07378","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x4d-sceneformer-enhanced-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2312.07378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07378"}},"official":{"repos":["jinglinglingling/x4d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-well-does-gpt-4v-ision-adapt-to","slug":"how-well-does-gpt-4v-ision-adapt-to","title":"How Well Does GPT-4V(ision) Adapt to Distribution Shifts? A Preliminary Investigation","date":"2023-12-12","arxiv_id":"2312.07424","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-well-does-gpt-4v-ision-adapt-to#ran","syntology_url":"https://syntology.ai/paper/2312.07424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07424"}},"official":{"repos":["jameszhou-gl/gpt-4v-distribution-shift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-question-answering-in-long","slug":"grounded-question-answering-in-long","title":"Grounded Question-Answering in Long Egocentric Videos","date":"2023-12-11","arxiv_id":"2312.06505","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-question-answering-in-long#ran","syntology_url":"https://syntology.ai/paper/2312.06505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06505"}},"official":{"repos":["becomebright/groundvqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/action-scene-graphs-for-long-form","slug":"action-scene-graphs-for-long-form","title":"Action Scene Graphs for Long-Form Understanding of Egocentric Videos","date":"2023-12-06","arxiv_id":"2312.03391","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-scene-graphs-for-long-form#ran","syntology_url":"https://syntology.ai/paper/2312.03391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03391"}},"official":{"repos":["fpv-iplab/easg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timechat-a-time-sensitive-multimodal-large","slug":"timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","arxiv_id":"2312.02051","repositories_listed":2,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/timechat-a-time-sensitive-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2312.02051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02051"}},"official":{"repos":["renshuhuai-andy/timechat"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/ego-exo4d-understanding-skilled-human","slug":"ego-exo4d-understanding-skilled-human","title":"Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives","date":"2023-11-30","arxiv_id":"2311.18259","repositories_listed":2,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/ego-exo4d-understanding-skilled-human#ran","syntology_url":"https://syntology.ai/paper/2311.18259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18259"}},"official":{"repos":["facebookresearch/Ego4d"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cast-cross-attention-in-space-and-time-for-1","slug":"cast-cross-attention-in-space-and-time-for-1","title":"CAST: Cross-Attention in Space and Time for Video Action Recognition","date":"2023-11-30","arxiv_id":"2311.18825","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cast-cross-attention-in-space-and-time-for-1#ran","syntology_url":"https://syntology.ai/paper/2311.18825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18825"}},"official":null}},{"url":"/paper/mvbench-a-comprehensive-multi-modal-video","slug":"mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","arxiv_id":"2311.17005","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mvbench-a-comprehensive-multi-modal-video#ran","syntology_url":"https://syntology.ai/paper/2311.17005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17005"}},"official":{"repos":["opengvlab/ask-anything"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/panoptic-video-scene-graph-generation-1","slug":"panoptic-video-scene-graph-generation-1","title":"Panoptic Video Scene Graph Generation","date":"2023-11-28","arxiv_id":"2311.17058","repositories_listed":3,"syntology":{"n":11,"n_ran":11,"n_constructed":2,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/panoptic-video-scene-graph-generation-1#ran","syntology_url":"https://syntology.ai/paper/2311.17058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17058"}},"official":{"repos":["jingkang50/openpvsg","lilydaytoy/openpvsg","lilydaytoy/pvsgannotation"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":2,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pg-video-llava-pixel-grounding-large-video","slug":"pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","arxiv_id":"2311.13435","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pg-video-llava-pixel-grounding-large-video#ran","syntology_url":"https://syntology.ai/paper/2311.13435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13435"}},"official":{"repos":["mbzuai-oryx/video-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vamos-versatile-action-models-for-video","slug":"vamos-versatile-action-models-for-video","title":"Vamos: Versatile Action Models for Video Understanding","date":"2023-11-22","arxiv_id":"2311.13627","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vamos-versatile-action-models-for-video#ran","syntology_url":"https://syntology.ai/paper/2311.13627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13627"}},"official":{"repos":["brown-palm/Vamos"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/one-for-all-video-conversation-is-feasible","slug":"one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","arxiv_id":"2309.15785","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/one-for-all-video-conversation-is-feasible#ran","syntology_url":"https://syntology.ai/paper/2309.15785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15785"}},"official":{"repos":["farewellthree/BT-Adapter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cefhri-a-communication-efficient-federated","slug":"cefhri-a-communication-efficient-federated","title":"CEFHRI: A Communication Efficient Federated Learning Framework for Recognizing Industrial Human-Robot Interaction","date":"2023-08-29","arxiv_id":"2308.14965","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cefhri-a-communication-efficient-federated#ran","syntology_url":"https://syntology.ai/paper/2308.14965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14965"}},"official":{"repos":["umarkhalidai/cefhri-efficient-federated-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mofo-motion-focused-self-supervision-for","slug":"mofo-motion-focused-self-supervision-for","title":"MOFO: MOtion FOcused Self-Supervision for Video Understanding","date":"2023-08-23","arxiv_id":"2308.12447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mofo-motion-focused-self-supervision-for#ran","syntology_url":"https://syntology.ai/paper/2308.12447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12447"}},"official":{"repos":["moohnai/mofo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-current-long-term-video-understanding","slug":"are-current-long-term-video-understanding","title":"Are current long-term video understanding datasets long-term?","date":"2023-08-22","arxiv_id":"2308.11244","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/are-current-long-term-video-understanding#ran","syntology_url":"https://syntology.ai/paper/2308.11244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11244"}},"official":{"repos":["ombretta/longterm_datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-spatio-temporal-structure-prediction","slug":"masked-spatio-temporal-structure-prediction","title":"Masked Spatio-Temporal Structure Prediction for Self-supervised Learning on Point Cloud Videos","date":"2023-08-18","arxiv_id":"2308.09245","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/masked-spatio-temporal-structure-prediction#ran","syntology_url":"https://syntology.ai/paper/2308.09245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09245"}},"official":{"repos":["johnsonsign/mast-pre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egoschema-a-diagnostic-benchmark-for-very-1#ran","syntology_url":"https://syntology.ai/paper/2308.09126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09126"}},"official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/helping-hands-an-object-aware-ego-centric","slug":"helping-hands-an-object-aware-ego-centric","title":"Helping Hands: An Object-Aware Ego-Centric Video Recognition Model","date":"2023-08-15","arxiv_id":"2308.07918","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/helping-hands-an-object-aware-ego-centric#ran","syntology_url":"https://syntology.ai/paper/2308.07918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07918"}},"official":{"repos":["chuhanxx/helping_hand_for_egocentric_videos"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-adaptive-models-for-efficient","slug":"temporally-adaptive-models-for-efficient","title":"Temporally-Adaptive Models for Efficient Video Understanding","date":"2023-08-10","arxiv_id":"2308.05787","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporally-adaptive-models-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2308.05787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05787"}},"official":{"repos":["alibaba-mmai-research/TAdaConv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-distillation-for-egocentric-action","slug":"multimodal-distillation-for-egocentric-action","title":"Multimodal Distillation for Egocentric Action Recognition","date":"2023-07-14","arxiv_id":"2307.07483","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-distillation-for-egocentric-action#ran","syntology_url":"https://syntology.ai/paper/2307.07483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07483"}},"official":{"repos":["gorjanradevski/multimodal-distillation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ha-vid-a-human-assembly-video-dataset-for","slug":"ha-vid-a-human-assembly-video-dataset-for","title":"HA-ViD: A Human Assembly Video Dataset for Comprehensive Assembly Knowledge Understanding","date":"2023-07-09","arxiv_id":"2307.05721","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ha-vid-a-human-assembly-video-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2307.05721","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05721"}},"official":{"repos":["iai-hrc/ha-vid"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/youku-mplug-a-10-million-large-scale-chinese","slug":"youku-mplug-a-10-million-large-scale-chinese","title":"Youku-mPLUG: A 10 Million Large-scale Chinese Video-Language Dataset for Pre-training and Benchmarks","date":"2023-06-07","arxiv_id":"2306.04362","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/youku-mplug-a-10-million-large-scale-chinese#ran","syntology_url":"https://syntology.ai/paper/2306.04362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04362"}},"official":{"repos":["x-plug/youku-mplug"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/video-llama-an-instruction-tuned-audio-visual","slug":"video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","arxiv_id":"2306.02858","repositories_listed":4,"syntology":{"n":25,"n_ran":18,"n_constructed":7,"n_ran_checked":8,"n_instrument":10,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":9,"phrase":"18 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 10 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/video-llama-an-instruction-tuned-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2306.02858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02858"}},"official":{"repos":["damo-nlp-sg/video-llama"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/procedure-aware-pretraining-for-instructional","slug":"procedure-aware-pretraining-for-instructional","title":"Procedure-Aware Pretraining for Instructional Video Understanding","date":"2023-03-31","arxiv_id":"2303.18230","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/procedure-aware-pretraining-for-instructional#ran","syntology_url":"https://syntology.ai/paper/2303.18230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.18230"}},"official":{"repos":["salesforce/paprika"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/query-dependent-video-representation-for","slug":"query-dependent-video-representation-for","title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","date":"2023-03-24","arxiv_id":"2303.13874","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/query-dependent-video-representation-for#ran","syntology_url":"https://syntology.ai/paper/2303.13874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13874"}},"official":{"repos":["wjun0830/qd-detr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-video-representation","slug":"weakly-supervised-video-representation","title":"Weakly Supervised Video Representation Learning with Unaligned Text for Sequential Videos","date":"2023-03-22","arxiv_id":"2303.12370","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":2,"n_instrument":6,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/weakly-supervised-video-representation#ran","syntology_url":"https://syntology.ai/paper/2303.12370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12370"}},"official":{"repos":["svip-lab/weaksvr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/test-of-time-instilling-video-language-models","slug":"test-of-time-instilling-video-language-models","title":"Test of Time: Instilling Video-Language Models with a Sense of Time","date":"2023-01-05","arxiv_id":"2301.02074","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/test-of-time-instilling-video-language-models#ran","syntology_url":"https://syntology.ai/paper/2301.02074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02074"}},"official":{"repos":["bpiyush/TestOfTime"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/internvideo-general-video-foundation-models","slug":"internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","arxiv_id":"2212.03191","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internvideo-general-video-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2212.03191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03191"}},"official":{"repos":["opengvlab/internvideo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-video-representation-learning-via","slug":"efficient-video-representation-learning-via","title":"EVEREST: Efficient Masked Video Autoencoder by Removing Redundant Spatiotemporal Tokens","date":"2022-11-19","arxiv_id":"2211.10636","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-video-representation-learning-via#ran","syntology_url":"https://syntology.ai/paper/2211.10636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10636"}},"official":{"repos":["sunilhoho/everest","sunilhoho/VideoMS"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/uniformerv2-spatiotemporal-learning-by-arming-1","slug":"uniformerv2-spatiotemporal-learning-by-arming-1","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-11-17","arxiv_id":"2211.09552","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/uniformerv2-spatiotemporal-learning-by-arming-1#ran","syntology_url":"https://syntology.ai/paper/2211.09552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09552"}},"official":{"repos":["OpenGVLab/UniFormerV2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-action-segmentation-an-analysis-of","slug":"temporal-action-segmentation-an-analysis-of","title":"Temporal Action Segmentation: An Analysis of Modern Techniques","date":"2022-10-19","arxiv_id":"2210.10352","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/temporal-action-segmentation-an-analysis-of#ran","syntology_url":"https://syntology.ai/paper/2210.10352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10352"}},"official":{"repos":["atlas-eccv22/awesome-temporal-action-segmentation","nus-cvml/awesome-temporal-action-segmentation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vtc-improving-video-text-retrieval-with-user","slug":"vtc-improving-video-text-retrieval-with-user","title":"VTC: Improving Video-Text Retrieval with User Comments","date":"2022-10-19","arxiv_id":"2210.10820","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtc-improving-video-text-retrieval-with-user#ran","syntology_url":"https://syntology.ai/paper/2210.10820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10820"}},"official":null}},{"url":"/paper/how-would-the-viewer-feel-estimating","slug":"how-would-the-viewer-feel-estimating","title":"How Would The Viewer Feel? Estimating Wellbeing From Video Scenarios","date":"2022-10-18","arxiv_id":"2210.10039","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-would-the-viewer-feel-estimating#ran","syntology_url":"https://syntology.ai/paper/2210.10039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10039"}},"official":{"repos":["hendrycks/emodiversity"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egotaskqa-understanding-human-tasks-in","slug":"egotaskqa-understanding-human-tasks-in","title":"EgoTaskQA: Understanding Human Tasks in Egocentric Videos","date":"2022-10-08","arxiv_id":"2210.03929","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egotaskqa-understanding-human-tasks-in#ran","syntology_url":"https://syntology.ai/paper/2210.03929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03929"}},"official":{"repos":["Buzz-Beater/EgoTaskQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/panoramic-vision-transformer-for-saliency","slug":"panoramic-vision-transformer-for-saliency","title":"Panoramic Vision Transformer for Saliency Detection in 360° Videos","date":"2022-09-19","arxiv_id":"2209.08956","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/panoramic-vision-transformer-for-saliency#ran","syntology_url":"https://syntology.ai/paper/2209.08956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08956"}},"official":{"repos":["hs-yn/paver"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/static-and-dynamic-concepts-for-self","slug":"static-and-dynamic-concepts-for-self","title":"Static and Dynamic Concepts for Self-supervised Video Representation Learning","date":"2022-07-26","arxiv_id":"2207.12795","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/static-and-dynamic-concepts-for-self#ran","syntology_url":"https://syntology.ai/paper/2207.12795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12795"}},"official":{"repos":["shvdiwnkozbw/Self-supervised-Video-Concept"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/spotting-temporally-precise-fine-grained","slug":"spotting-temporally-precise-fine-grained","title":"Spotting Temporally Precise, Fine-Grained Events in Video","date":"2022-07-20","arxiv_id":"2207.10213","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spotting-temporally-precise-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2207.10213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10213"}},"official":{"repos":["soccernet/sn-spotting","jhong93/spot"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-image-to-video-transfer","slug":"parameter-efficient-image-to-video-transfer","title":"ST-Adapter: Parameter-Efficient Image-to-Video Transfer Learning","date":"2022-06-27","arxiv_id":"2206.13559","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/parameter-efficient-image-to-video-transfer#ran","syntology_url":"https://syntology.ai/paper/2206.13559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13559"}},"official":{"repos":["linziyi96/st-adapter"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-representation-to-reasoning-towards-both","slug":"from-representation-to-reasoning-towards-both","title":"From Representation to Reasoning: Towards both Evidence and Commonsense Reasoning for Video Question-Answering","date":"2022-05-30","arxiv_id":"2205.14895","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-representation-to-reasoning-towards-both#ran","syntology_url":"https://syntology.ai/paper/2205.14895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14895"}},"official":{"repos":["bcmi/causal-vidqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","repositories_listed":5,"syntology":{"n":24,"n_ran":18,"n_constructed":6,"n_ran_checked":12,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":8,"phrase":"18 ran (of which 6 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}},"official":null}},{"url":"/paper/an-empirical-study-of-end-to-end-temporal","slug":"an-empirical-study-of-end-to-end-temporal","title":"An Empirical Study of End-to-End Temporal Action Detection","date":"2022-04-06","arxiv_id":"2204.02932","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-empirical-study-of-end-to-end-temporal#ran","syntology_url":"https://syntology.ai/paper/2204.02932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02932"}},"official":{"repos":["xlliu7/E2E-TAD"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-alignment-networks-for-long-term","slug":"temporal-alignment-networks-for-long-term","title":"Temporal Alignment Networks for Long-term Video","date":"2022-04-06","arxiv_id":"2204.02968","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/temporal-alignment-networks-for-long-term#ran","syntology_url":"https://syntology.ai/paper/2204.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02968"}},"official":null}},{"url":"/paper/long-movie-clip-classification-with-state","slug":"long-movie-clip-classification-with-state","title":"Long Movie Clip Classification with State-Space Video Models","date":"2022-04-04","arxiv_id":"2204.01692","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":2,"n_ran_checked":3,"n_instrument":8,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/long-movie-clip-classification-with-state#ran","syntology_url":"https://syntology.ai/paper/2204.01692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01692"}},"official":{"repos":["md-mohaiminul/ViS4mer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/spact-self-supervised-privacy-preservation","slug":"spact-self-supervised-privacy-preservation","title":"SPAct: Self-supervised Privacy Preservation for Action Recognition","date":"2022-03-29","arxiv_id":"2203.15205","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/spact-self-supervised-privacy-preservation#ran","syntology_url":"https://syntology.ai/paper/2203.15205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15205"}},"official":{"repos":["daveishan/spact"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fitclip-refining-large-scale-pretrained-image","slug":"fitclip-refining-large-scale-pretrained-image","title":"FitCLIP: Refining Large-Scale Pretrained Image-Text Models for Zero-Shot Video Understanding Tasks","date":"2022-03-24","arxiv_id":"2203.13371","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fitclip-refining-large-scale-pretrained-image#ran","syntology_url":"https://syntology.ai/paper/2203.13371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13371"}},"official":{"repos":["bryant1410/fitclip","bryant1410/tclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videomae-masked-autoencoders-are-data-1","slug":"videomae-masked-autoencoders-are-data-1","title":"VideoMAE: Masked Autoencoders are Data-Efficient Learners for Self-Supervised Video Pre-Training","date":"2022-03-23","arxiv_id":"2203.12602","repositories_listed":9,"syntology":{"n":13,"n_ran":10,"n_constructed":6,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":12,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/videomae-masked-autoencoders-are-data-1#ran","syntology_url":"https://syntology.ai/paper/2203.12602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.12602"}},"official":{"repos":["MCG-NJU/VideoMAE","MCG-NJU/VideoMAE-Action-Detection"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-optical-flow-with-adaptive-graph","slug":"learning-optical-flow-with-adaptive-graph","title":"Learning Optical Flow with Adaptive Graph Reasoning","date":"2022-02-08","arxiv_id":"2202.03857","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-optical-flow-with-adaptive-graph#ran","syntology_url":"https://syntology.ai/paper/2202.03857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03857"}},"official":{"repos":["la30/agflow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-dataset-for-medical-instructional-video","slug":"a-dataset-for-medical-instructional-video","title":"A Dataset for Medical Instructional Video Classification and Question Answering","date":"2022-01-30","arxiv_id":"2201.12888","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-dataset-for-medical-instructional-video#ran","syntology_url":"https://syntology.ai/paper/2201.12888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12888"}},"official":{"repos":["deepaknlp/medvidqacl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/capturing-temporal-information-in-a-single","slug":"capturing-temporal-information-in-a-single","title":"Capturing Temporal Information in a Single Frame: Channel Sampling Strategies for Action Recognition","date":"2022-01-25","arxiv_id":"2201.10394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/capturing-temporal-information-in-a-single#ran","syntology_url":"https://syntology.ai/paper/2201.10394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.10394"}},"official":{"repos":["kiyoon/channel_sampling"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-visual-language-models-for","slug":"prompting-visual-language-models-for","title":"Prompting Visual-Language Models for Efficient Video Understanding","date":"2021-12-08","arxiv_id":"2112.04478","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prompting-visual-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2112.04478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.04478"}},"official":{"repos":["ju-chen/Efficient-Prompt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-referring-video-object","slug":"end-to-end-referring-video-object","title":"End-to-End Referring Video Object Segmentation with Multimodal Transformers","date":"2021-11-29","arxiv_id":"2111.14821","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-referring-video-object#ran","syntology_url":"https://syntology.ai/paper/2111.14821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14821"}},"official":{"repos":["mttr2021/MTTR"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/swinbert-end-to-end-transformers-with-sparse","slug":"swinbert-end-to-end-transformers-with-sparse","title":"SwinBERT: End-to-End Transformers with Sparse Attention for Video Captioning","date":"2021-11-25","arxiv_id":"2111.13196","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swinbert-end-to-end-transformers-with-sparse#ran","syntology_url":"https://syntology.ai/paper/2111.13196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13196"}},"official":{"repos":["microsoft/swinbert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-pyramid-multimodal-pyramid-attentional","slug":"mm-pyramid-multimodal-pyramid-attentional","title":"MM-Pyramid: Multimodal Pyramid Attentional Network for Audio-Visual Event Localization and Video Parsing","date":"2021-11-24","arxiv_id":"2111.12374","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-pyramid-multimodal-pyramid-attentional#ran","syntology_url":"https://syntology.ai/paper/2111.12374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12374"}},"official":{"repos":["JustinYuu/MM_Pyramid"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pytorchvideo-a-deep-learning-library-for","slug":"pytorchvideo-a-deep-learning-library-for","title":"PyTorchVideo: A Deep Learning Library for Video Understanding","date":"2021-11-18","arxiv_id":"2111.09887","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pytorchvideo-a-deep-learning-library-for#ran","syntology_url":"https://syntology.ai/paper/2111.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.09887"}},"official":{"repos":["facebookresearch/pytorchvideo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/relational-self-attention-what-s-missing-in","slug":"relational-self-attention-what-s-missing-in","title":"Relational Self-Attention: What's Missing in Attention for Video Understanding","date":"2021-11-02","arxiv_id":"2111.01673","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/relational-self-attention-what-s-missing-in#ran","syntology_url":"https://syntology.ai/paper/2111.01673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.01673"}},"official":{"repos":["KimManjin/RSA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/object-region-video-transformers-1","slug":"object-region-video-transformers-1","title":"Object-Region Video Transformers","date":"2021-10-13","arxiv_id":"2110.06915","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/object-region-video-transformers-1#ran","syntology_url":"https://syntology.ai/paper/2110.06915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06915"}},"official":null}},{"url":"/paper/learning-temporally-causal-latent-processes","slug":"learning-temporally-causal-latent-processes","title":"Learning Temporally Causal Latent Processes from General Temporal Data","date":"2021-10-11","arxiv_id":"2110.05428","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-temporally-causal-latent-processes#ran","syntology_url":"https://syntology.ai/paper/2110.05428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.05428"}},"official":{"repos":["weirayao/leap"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","listed"]}}},{"url":"/paper/token-shift-transformer-for-video","slug":"token-shift-transformer-for-video","title":"Token Shift Transformer for Video Classification","date":"2021-08-05","arxiv_id":"2108.02432","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/token-shift-transformer-for-video#ran","syntology_url":"https://syntology.ai/paper/2108.02432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02432"}},"official":{"repos":["VideoNetworks/TokShift-Transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/elaborative-rehearsal-for-zero-shot-action","slug":"elaborative-rehearsal-for-zero-shot-action","title":"Elaborative Rehearsal for Zero-shot Action Recognition","date":"2021-08-05","arxiv_id":"2108.02833","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/elaborative-rehearsal-for-zero-shot-action#ran","syntology_url":"https://syntology.ai/paper/2108.02833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02833"}},"official":{"repos":["DeLightCMU/ElaborativeRehearsal"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-self-supervised-video","slug":"enhancing-self-supervised-video","title":"Enhancing Self-supervised Video Representation Learning via Multi-level Feature Optimization","date":"2021-08-04","arxiv_id":"2108.02183","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-self-supervised-video#ran","syntology_url":"https://syntology.ai/paper/2108.02183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02183"}},"official":{"repos":["shvdiwnkozbw/video-representation-via-multi-level-optimization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/spatial-temporal-transformer-for-dynamic","slug":"spatial-temporal-transformer-for-dynamic","title":"Spatial-Temporal Transformer for Dynamic Scene Graph Generation","date":"2021-07-26","arxiv_id":"2107.12309","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":4,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":6,"phrase":"13 ran (of which 4 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spatial-temporal-transformer-for-dynamic#ran","syntology_url":"https://syntology.ai/paper/2107.12309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.12309"}},"official":{"repos":["yrcong/sttran"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/an-image-classifier-can-suffice-video","slug":"an-image-classifier-can-suffice-video","title":"Can An Image Classifier Suffice For Action Recognition?","date":"2021-06-26","arxiv_id":"2106.14104","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-image-classifier-can-suffice-video#ran","syntology_url":"https://syntology.ai/paper/2106.14104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14104"}},"official":{"repos":["ibm/sifar-pytorch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/video-swin-transformer","slug":"video-swin-transformer","title":"Video Swin Transformer","date":"2021-06-24","arxiv_id":"2106.13230","repositories_listed":15,"syntology":{"n":32,"n_ran":19,"n_constructed":0,"n_ran_checked":14,"n_instrument":5,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":7,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 5 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/video-swin-transformer#ran","syntology_url":"https://syntology.ai/paper/2106.13230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13230"}},"official":{"repos":["SwinTransformer/Video-Swin-Transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/vimpac-video-pre-training-via-masked-token","slug":"vimpac-video-pre-training-via-masked-token","title":"VIMPAC: Video Pre-Training via Masked Token Prediction and Contrastive Learning","date":"2021-06-21","arxiv_id":"2106.11250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vimpac-video-pre-training-via-masked-token#ran","syntology_url":"https://syntology.ai/paper/2106.11250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11250"}},"official":{"repos":["airsplay/vimpac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tokenlearner-what-can-8-learned-tokens-do-for","slug":"tokenlearner-what-can-8-learned-tokens-do-for","title":"TokenLearner: What Can 8 Learned Tokens Do for Images and Videos?","date":"2021-06-21","arxiv_id":"2106.11297","repositories_listed":11,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tokenlearner-what-can-8-learned-tokens-do-for#ran","syntology_url":"https://syntology.ai/paper/2106.11297","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11297"}},"official":{"repos":["google-research/scenic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-long-form-video-understanding-1","slug":"towards-long-form-video-understanding-1","title":"Towards Long-Form Video Understanding","date":"2021-06-21","arxiv_id":"2106.11310","repositories_listed":2,"syntology":{"n":19,"n_ran":18,"n_constructed":0,"n_ran_checked":16,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":14,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-long-form-video-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2106.11310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11310"}},"official":{"repos":["chaoyuaw/lvu"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-temporal-action-detection-with","slug":"end-to-end-temporal-action-detection-with","title":"End-to-end Temporal Action Detection with Transformer","date":"2021-06-18","arxiv_id":"2106.10271","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-temporal-action-detection-with#ran","syntology_url":"https://syntology.ai/paper/2106.10271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.10271"}},"official":{"repos":["xlliu7/TadTR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/next-qa-next-phase-of-question-answering-to","slug":"next-qa-next-phase-of-question-answering-to","title":"NExT-QA:Next Phase of Question-Answering to Explaining Temporal Actions","date":"2021-05-18","arxiv_id":"2105.08276","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/next-qa-next-phase-of-question-answering-to#ran","syntology_url":"https://syntology.ai/paper/2105.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08276"}},"official":{"repos":["doc-doc/NExT-QA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stochastic-image-to-video-synthesis-using","slug":"stochastic-image-to-video-synthesis-using","title":"Stochastic Image-to-Video Synthesis using cINNs","date":"2021-05-10","arxiv_id":"2105.04551","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":4,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stochastic-image-to-video-synthesis-using#ran","syntology_url":"https://syntology.ai/paper/2105.04551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.04551"}},"official":{"repos":["CompVis/image2video-synthesis-using-cINNs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/frameexit-conditional-early-exiting-for","slug":"frameexit-conditional-early-exiting-for","title":"FrameExit: Conditional Early Exiting for Efficient Video Recognition","date":"2021-04-27","arxiv_id":"2104.13400","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":1,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/frameexit-conditional-early-exiting-for#ran","syntology_url":"https://syntology.ai/paper/2104.13400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.13400"}},"official":{"repos":["Qualcomm-AI-research/FrameExit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/clip4clip-an-empirical-study-of-clip-for-end","slug":"clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","arxiv_id":"2104.08860","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clip4clip-an-empirical-study-of-clip-for-end#ran","syntology_url":"https://syntology.ai/paper/2104.08860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08860"}},"official":{"repos":["ArrowLuo/CLIP4Clip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/tuber-tube-transformer-for-action-detection","slug":"tuber-tube-transformer-for-action-detection","title":"TubeR: Tubelet Transformer for Video Action Detection","date":"2021-04-02","arxiv_id":"2104.00969","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tuber-tube-transformer-for-action-detection#ran","syntology_url":"https://syntology.ai/paper/2104.00969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00969"}},"official":null}},{"url":"/paper/visual-semantic-role-labeling-for-video","slug":"visual-semantic-role-labeling-for-video","title":"Visual Semantic Role Labeling for Video Understanding","date":"2021-04-02","arxiv_id":"2104.00990","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-semantic-role-labeling-for-video#ran","syntology_url":"https://syntology.ai/paper/2104.00990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00990"}},"official":{"repos":["TheShadow29/VidSitu"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-weighted-hierarchical-clustering","slug":"temporally-weighted-hierarchical-clustering","title":"Temporally-Weighted Hierarchical Clustering for Unsupervised Action Segmentation","date":"2021-03-20","arxiv_id":"2103.11264","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/temporally-weighted-hierarchical-clustering#ran","syntology_url":"https://syntology.ai/paper/2103.11264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.11264"}},"official":{"repos":["ssarfraz/FINCH-CLustering"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"34127f7d4db441a19f063e7ca541fc71ed338905815a990b72a61f755f630d68","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}