{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-question-answering/papers/ran/1","list_of":"/task/video-question-answering","task":"Video Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":124,"counts":{"archive_papers_tagged":460,"with_a_code_link":250,"where_syntology_ran_a_sample":124,"not_listed_spam_title":0,"listed":460,"listed_where_code_ran":124,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":107,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":107,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-question-answering/papers/ran/1","prev":null,"next":"/task/video-question-answering/papers/ran/2","papers":[{"url":"/paper/llava-scissor-token-compression-with-semantic","slug":"llava-scissor-token-compression-with-semantic","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","date":"2025-06-27","arxiv_id":"2506.21862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llava-scissor-token-compression-with-semantic#ran","syntology_url":"https://syntology.ai/paper/2506.21862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21862"}},"official":{"repos":["HumanMLLM/LLaVA-Scissor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/v-jepa-2-self-supervised-video-models-enable","slug":"v-jepa-2-self-supervised-video-models-enable","title":"V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning","date":"2025-06-11","arxiv_id":"2506.09985","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/v-jepa-2-self-supervised-video-models-enable#ran","syntology_url":"https://syntology.ai/paper/2506.09985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09985"}},"official":{"repos":["facebookresearch/vjepa2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/agentic-keyframe-search-for-video-question","slug":"agentic-keyframe-search-for-video-question","title":"Agentic Keyframe Search for Video Question Answering","date":"2025-03-20","arxiv_id":"2503.16032","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-keyframe-search-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2503.16032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16032"}},"official":{"repos":["fansunqi/akeys"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/videomind-a-chain-of-lora-agent-for-long","slug":"videomind-a-chain-of-lora-agent-for-long","title":"VideoMind: A Chain-of-LoRA Agent for Long Video Reasoning","date":"2025-03-17","arxiv_id":"2503.13444","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomind-a-chain-of-lora-agent-for-long#ran","syntology_url":"https://syntology.ai/paper/2503.13444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13444"}},"official":{"repos":["yeliudev/VideoMind"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-video-question-answering-with-in","slug":"streaming-video-question-answering-with-in","title":"Streaming Video Question-Answering with In-context Video KV-Cache Retrieval","date":"2025-03-01","arxiv_id":"2503.00540","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/streaming-video-question-answering-with-in#ran","syntology_url":"https://syntology.ai/paper/2503.00540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00540"}},"official":{"repos":["becomebright/rekv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/egotextvqa-towards-egocentric-scene-text","slug":"egotextvqa-towards-egocentric-scene-text","title":"EgoTextVQA: Towards Egocentric Scene-Text Aware Video Question Answering","date":"2025-02-11","arxiv_id":"2502.07411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egotextvqa-towards-egocentric-scene-text#ran","syntology_url":"https://syntology.ai/paper/2502.07411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07411"}},"official":{"repos":["zhousheng97/egotextvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infty-video-a-training-free-approach-to-long","slug":"infty-video-a-training-free-approach-to-long","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","date":"2025-01-31","arxiv_id":"2501.19098","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infty-video-a-training-free-approach-to-long#ran","syntology_url":"https://syntology.ai/paper/2501.19098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19098"}},"official":{"repos":["deep-spin/infinite-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-3-frontier-multimodal-foundation","slug":"videollama-3-frontier-multimodal-foundation","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","date":"2025-01-22","arxiv_id":"2501.13106","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/videollama-3-frontier-multimodal-foundation#ran","syntology_url":"https://syntology.ai/paper/2501.13106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13106"}},"official":{"repos":["damo-nlp-sg/videollama3"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/ecbench-can-multi-modal-foundation-models","slug":"ecbench-can-multi-modal-foundation-models","title":"ECBench: Can Multi-modal Foundation Models Understand the Egocentric World? A Holistic Embodied Cognition Benchmark","date":"2025-01-09","arxiv_id":"2501.05031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ecbench-can-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2501.05031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05031"}},"official":{"repos":["rh-dang/ecbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/linvt-empower-your-image-level-large-language","slug":"linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","arxiv_id":"2412.05185","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":6,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/linvt-empower-your-image-level-large-language#ran","syntology_url":"https://syntology.ai/paper/2412.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05185"}},"official":{"repos":["gls0425/linvt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/videollm-knows-when-to-speak-enhancing-time","slug":"videollm-knows-when-to-speak-enhancing-time","title":"VideoLLM Knows When to Speak: Enhancing Time-Sensitive Video Comprehension with Video-Text Duet Interaction Format","date":"2024-11-27","arxiv_id":"2411.17991","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videollm-knows-when-to-speak-enhancing-time#ran","syntology_url":"https://syntology.ai/paper/2411.17991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17991"}},"official":{"repos":["yellow-binary-tree/mmduet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videoespresso-a-large-scale-chain-of-thought","slug":"videoespresso-a-large-scale-chain-of-thought","title":"VideoEspresso: A Large-Scale Chain-of-Thought Dataset for Fine-Grained Video Reasoning via Core Frame Selection","date":"2024-11-22","arxiv_id":"2411.14794","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videoespresso-a-large-scale-chain-of-thought#ran","syntology_url":"https://syntology.ai/paper/2411.14794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14794"}},"official":{"repos":["hshjerry/videoespresso"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ppllava-varied-video-sequence-understanding","slug":"ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ppllava-varied-video-sequence-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.02327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02327"}},"official":{"repos":["farewellthree/ppllava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/timesuite-improving-mllms-for-long-video","slug":"timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","arxiv_id":"2410.19702","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timesuite-improving-mllms-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2410.19702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19702"}},"official":null}},{"url":"/paper/longvu-spatiotemporal-adaptive-compression","slug":"longvu-spatiotemporal-adaptive-compression","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","date":"2024-10-22","arxiv_id":"2410.17434","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvu-spatiotemporal-adaptive-compression#ran","syntology_url":"https://syntology.ai/paper/2410.17434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17434"}},"official":{"repos":["Vision-CAIR/LongVU"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videgothink-assessing-egocentric-video","slug":"videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","arxiv_id":"2410.11623","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":5,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videgothink-assessing-egocentric-video#ran","syntology_url":"https://syntology.ai/paper/2410.11623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11623"}},"official":null}},{"url":"/paper/temporalbench-benchmarking-fine-grained","slug":"temporalbench-benchmarking-fine-grained","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","date":"2024-10-14","arxiv_id":"2410.10818","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporalbench-benchmarking-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.10818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10818"}},"official":{"repos":["mu-cai/TemporalBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-temporal-modeling-of-video-llms-via","slug":"enhancing-temporal-modeling-of-video-llms-via","title":"Enhancing Temporal Modeling of Video LLMs via Time Gating","date":"2024-10-08","arxiv_id":"2410.05714","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/enhancing-temporal-modeling-of-video-llms-via#ran","syntology_url":"https://syntology.ai/paper/2410.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05714"}},"official":{"repos":["lavi-lab/tg-vid"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/aria-an-open-multimodal-native-mixture-of","slug":"aria-an-open-multimodal-native-mixture-of","title":"Aria: An Open Multimodal Native Mixture-of-Experts Model","date":"2024-10-08","arxiv_id":"2410.05993","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aria-an-open-multimodal-native-mixture-of#ran","syntology_url":"https://syntology.ai/paper/2410.05993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05993"}},"official":{"repos":["rhymes-ai/aria"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/remembr-building-and-reasoning-over-long","slug":"remembr-building-and-reasoning-over-long","title":"ReMEmbR: Building and Reasoning Over Long-Horizon Spatio-Temporal Memory for Robot Navigation","date":"2024-09-20","arxiv_id":"2409.13682","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/remembr-building-and-reasoning-over-long#ran","syntology_url":"https://syntology.ai/paper/2409.13682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13682"}},"official":{"repos":["NVIDIA-AI-IOT/remembr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/oryx-mllm-on-demand-spatial-temporal","slug":"oryx-mllm-on-demand-spatial-temporal","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","date":"2024-09-19","arxiv_id":"2409.12961","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/oryx-mllm-on-demand-spatial-temporal#ran","syntology_url":"https://syntology.ai/paper/2409.12961","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12961"}},"official":{"repos":["oryx-mllm/oryx"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/qwen2-vl-enhancing-vision-language-model-s","slug":"qwen2-vl-enhancing-vision-language-model-s","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","date":"2024-09-18","arxiv_id":"2409.12191","repositories_listed":8,"syntology":{"n":12,"n_ran":12,"n_constructed":1,"n_ran_checked":10,"n_instrument":2,"n_unverified":0,"n_honours":4,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 10 with no instrument failure: 4 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/qwen2-vl-enhancing-vision-language-model-s#ran","syntology_url":"https://syntology.ai/paper/2409.12191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12191"}},"official":{"repos":["qwenlm/qwen2-vl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/grounded-multi-hop-videoqa-in-long-form","slug":"grounded-multi-hop-videoqa-in-long-form","title":"Grounded Multi-Hop VideoQA in Long-Form Egocentric Videos","date":"2024-08-26","arxiv_id":"2408.14469","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-multi-hop-videoqa-in-long-form#ran","syntology_url":"https://syntology.ai/paper/2408.14469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14469"}},"official":null}},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longvideobench-a-benchmark-for-long-context","slug":"longvideobench-a-benchmark-for-long-context","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","date":"2024-07-22","arxiv_id":"2407.15754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvideobench-a-benchmark-for-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.15754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15754"}},"official":{"repos":["longvideobench/longvideobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-next-interleave-tackling-multi-image","slug":"llava-next-interleave-tackling-multi-image","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","date":"2024-07-10","arxiv_id":"2407.07895","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llava-next-interleave-tackling-multi-image#ran","syntology_url":"https://syntology.ai/paper/2407.07895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07895"}},"official":{"repos":["LLaVA-VL/LLaVA-NeXT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/internlm-xcomposer-2-5-a-versatile-large","slug":"internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","arxiv_id":"2407.03320","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internlm-xcomposer-2-5-a-versatile-large#ran","syntology_url":"https://syntology.ai/paper/2407.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03320"}},"official":{"repos":["internlm/internlm-xcomposer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/tarsier-recipes-for-training-and-evaluating-1","slug":"tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","arxiv_id":"2407.00634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier-recipes-for-training-and-evaluating-1#ran","syntology_url":"https://syntology.ai/paper/2407.00634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00634"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/long-context-transfer-from-language-to-vision","slug":"long-context-transfer-from-language-to-vision","title":"Long Context Transfer from Language to Vision","date":"2024-06-24","arxiv_id":"2406.16852","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/long-context-transfer-from-language-to-vision#ran","syntology_url":"https://syntology.ai/paper/2406.16852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16852"}},"official":{"repos":["evolvinglmms-lab/longva"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/hcqa-ego4d-egoschema-challenge-2024","slug":"hcqa-ego4d-egoschema-challenge-2024","title":"HCQA @ Ego4D EgoSchema Challenge 2024","date":"2024-06-22","arxiv_id":"2406.15771","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hcqa-ego4d-egoschema-challenge-2024#ran","syntology_url":"https://syntology.ai/paper/2406.15771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15771"}},"official":{"repos":["hyu-zhang/hcqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voco-llama-towards-vision-compression-with","slug":"voco-llama-towards-vision-compression-with","title":"VoCo-LLaMA: Towards Vision Compression with Large Language Models","date":"2024-06-18","arxiv_id":"2406.12275","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":5,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voco-llama-towards-vision-compression-with#ran","syntology_url":"https://syntology.ai/paper/2406.12275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12275"}},"official":{"repos":["Yxxxb/VoCo-LLaMA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/i-srt-aligning-large-multimodal-models-for","slug":"i-srt-aligning-large-multimodal-models-for","title":"ISR-DPO: Aligning Large Multimodal Models for Videos by Iterative Self-Retrospective DPO","date":"2024-06-17","arxiv_id":"2406.11280","repositories_listed":4,"syntology":{"n":30,"n_ran":26,"n_constructed":0,"n_ran_checked":18,"n_instrument":8,"n_unverified":4,"n_honours":0,"n_violates":2,"n_no_contract":16,"n_pointer_only":15,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 2 violated, 16 with no contract checked; 8 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/i-srt-aligning-large-multimodal-models-for#ran","syntology_url":"https://syntology.ai/paper/2406.11280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11280"}},"official":{"repos":["snumprlab/SRT","snumprlab/isr-dpo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/too-many-frames-not-all-useful-efficient","slug":"too-many-frames-not-all-useful-efficient","title":"Too Many Frames, Not All Useful: Efficient Strategies for Long-Form Video QA","date":"2024-06-13","arxiv_id":"2406.09396","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/too-many-frames-not-all-useful-efficient#ran","syntology_url":"https://syntology.ai/paper/2406.09396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09396"}},"official":{"repos":["jongwoopark7978/LVNet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videogpt-integrating-image-and-video-encoders#ran","syntology_url":"https://syntology.ai/paper/2406.09418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09418"}},"official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-2-advancing-spatial-temporal","slug":"videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","arxiv_id":"2406.07476","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/videollama-2-advancing-spatial-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.07476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07476"}},"official":{"repos":["damo-nlp-sg/videollama2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/compositional-4d-dynamic-scenes-understanding","slug":"compositional-4d-dynamic-scenes-understanding","title":"Compositional 4D Dynamic Scenes Understanding with Physics Priors for Video Question Answering","date":"2024-06-02","arxiv_id":"2406.00622","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/compositional-4d-dynamic-scenes-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.00622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00622"}},"official":{"repos":["XingruiWang/SuperCLEVR-Physics"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/freeva-offline-mllm-as-training-free-video","slug":"freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","arxiv_id":"2405.07798","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/freeva-offline-mllm-as-training-free-video#ran","syntology_url":"https://syntology.ai/paper/2405.07798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07798"}},"official":{"repos":["whwu95/freeva"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/moviechat-question-aware-sparse-memory-for","slug":"moviechat-question-aware-sparse-memory-for","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","date":"2024-04-26","arxiv_id":"2404.17176","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/moviechat-question-aware-sparse-memory-for#ran","syntology_url":"https://syntology.ai/paper/2404.17176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17176"}},"official":{"repos":["rese1f/MovieChat"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt4-video-advancing-multimodal-llms-for","slug":"minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","arxiv_id":"2404.03413","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt4-video-advancing-multimodal-llms-for#ran","syntology_url":"https://syntology.ai/paper/2404.03413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03413"}},"official":{"repos":["Vision-CAIR/MiniGPT4-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-optimization-of-video-large","slug":"direct-preference-optimization-of-video-large","title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","date":"2024-04-01","arxiv_id":"2404.01258","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-preference-optimization-of-video-large#ran","syntology_url":"https://syntology.ai/paper/2404.01258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01258"}},"official":{"repos":["riflezhang/llava-hound-dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causalchaos-dataset-for-comprehensive-causal","slug":"causalchaos-dataset-for-comprehensive-causal","title":"CausalChaos! Dataset for Comprehensive Causal Action Question Answering Over Longer Causal Chains Grounded in Dynamic Visual Scenes","date":"2024-04-01","arxiv_id":"2404.01299","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causalchaos-dataset-for-comprehensive-causal#ran","syntology_url":"https://syntology.ai/paper/2404.01299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01299"}},"official":{"repos":["lunaproject22/causalchaos"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/st-llm-large-language-models-are-effective-1","slug":"st-llm-large-language-models-are-effective-1","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","date":"2024-03-30","arxiv_id":"2404.00308","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/st-llm-large-language-models-are-effective-1#ran","syntology_url":"https://syntology.ai/paper/2404.00308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00308"}},"official":{"repos":["TencentARC/ST-LLM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","slug":"an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","arxiv_id":"2403.18406","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-image-grid-can-be-worth-a-video-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2403.18406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18406"}},"official":{"repos":["imagegridworth/IG-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lita-language-instructed-temporal","slug":"lita-language-instructed-temporal","title":"LITA: Language Instructed Temporal-Localization Assistant","date":"2024-03-27","arxiv_id":"2403.19046","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lita-language-instructed-temporal#ran","syntology_url":"https://syntology.ai/paper/2403.19046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19046"}},"official":{"repos":["nvlabs/lita"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/omnivid-a-generative-framework-for-universal#ran","syntology_url":"https://syntology.ai/paper/2403.17935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17935"}},"official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/elysium-exploring-object-level-perception-in","slug":"elysium-exploring-object-level-perception-in","title":"Elysium: Exploring Object-level Perception in Videos via MLLM","date":"2024-03-25","arxiv_id":"2403.16558","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/elysium-exploring-object-level-perception-in#ran","syntology_url":"https://syntology.ai/paper/2403.16558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16558"}},"official":{"repos":["hon-wong/elysium"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vid-tldr-training-free-token-merging-for","slug":"vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","arxiv_id":"2403.13347","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vid-tldr-training-free-token-merging-for#ran","syntology_url":"https://syntology.ai/paper/2403.13347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13347"}},"official":{"repos":["mlvlab/vid-tldr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hawkeye-training-video-text-llms-for","slug":"hawkeye-training-video-text-llms-for","title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","date":"2024-03-15","arxiv_id":"2403.10228","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hawkeye-training-video-text-llms-for#ran","syntology_url":"https://syntology.ai/paper/2403.10228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.10228"}},"official":{"repos":["yellow-binary-tree/hawkeye"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lstp-language-guided-spatial-temporal-prompt","slug":"lstp-language-guided-spatial-temporal-prompt","title":"Efficient Temporal Extrapolation of Multimodal Large Language Models with Temporal Grounding Bridge","date":"2024-02-25","arxiv_id":"2402.16050","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lstp-language-guided-spatial-temporal-prompt#ran","syntology_url":"https://syntology.ai/paper/2402.16050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16050"}},"official":{"repos":["bigai-nlco/lstp-chat","bigai-nlco/videotgb"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crema-multimodal-compositional-video","slug":"crema-multimodal-compositional-video","title":"CREMA: Generalizable and Efficient Video-Language Reasoning via Multimodal Modular Fusion","date":"2024-02-08","arxiv_id":"2402.05889","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crema-multimodal-compositional-video#ran","syntology_url":"https://syntology.ai/paper/2402.05889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05889"}},"official":{"repos":["Yui010206/CREMA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glance-and-focus-memory-prompting-for-multi-1","slug":"glance-and-focus-memory-prompting-for-multi-1","title":"Glance and Focus: Memory Prompting for Multi-Event Video Question Answering","date":"2024-01-03","arxiv_id":"2401.01529","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":2,"n_ran_checked":14,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":13,"n_pointer_only":7,"phrase":"14 ran (of which 2 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/glance-and-focus-memory-prompting-for-multi-1#ran","syntology_url":"https://syntology.ai/paper/2401.01529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01529"}},"official":{"repos":["byz0e/glance-focus"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":2,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-llm-framework-for-long-range-video","slug":"a-simple-llm-framework-for-long-range-video","title":"A Simple LLM Framework for Long-Range Video Question-Answering","date":"2023-12-28","arxiv_id":"2312.17235","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-llm-framework-for-long-range-video#ran","syntology_url":"https://syntology.ai/paper/2312.17235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17235"}},"official":{"repos":["ceezh/llovi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lingoqa-video-question-answering-for","slug":"lingoqa-video-question-answering-for","title":"LingoQA: Visual Question Answering for Autonomous Driving","date":"2023-12-21","arxiv_id":"2312.14115","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lingoqa-video-question-answering-for#ran","syntology_url":"https://syntology.ai/paper/2312.14115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14115"}},"official":{"repos":["wayveai/lingoqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/shot2story20k-a-new-benchmark-for","slug":"shot2story20k-a-new-benchmark-for","title":"Shot2Story20K: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","date":"2023-12-16","arxiv_id":"2312.10300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shot2story20k-a-new-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2312.10300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10300"}},"official":{"repos":["bytedance/Shot2Story"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vlap-efficient-video-language-alignment-via","slug":"vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","arxiv_id":"2312.08367","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlap-efficient-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2312.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08367"}},"official":{"repos":["xijun-cs/vila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-question-answering-in-long","slug":"grounded-question-answering-in-long","title":"Grounded Question-Answering in Long Egocentric Videos","date":"2023-12-11","arxiv_id":"2312.06505","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-question-answering-in-long#ran","syntology_url":"https://syntology.ai/paper/2312.06505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06505"}},"official":{"repos":["becomebright/groundvqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timechat-a-time-sensitive-multimodal-large","slug":"timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","arxiv_id":"2312.02051","repositories_listed":2,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/timechat-a-time-sensitive-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2312.02051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02051"}},"official":{"repos":["renshuhuai-andy/timechat"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vtimellm-empower-llm-to-grasp-video-moments","slug":"vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","arxiv_id":"2311.18445","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vtimellm-empower-llm-to-grasp-video-moments#ran","syntology_url":"https://syntology.ai/paper/2311.18445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18445"}},"official":{"repos":["huangb23/vtimellm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mvbench-a-comprehensive-multi-modal-video","slug":"mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","arxiv_id":"2311.17005","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mvbench-a-comprehensive-multi-modal-video#ran","syntology_url":"https://syntology.ai/paper/2311.17005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17005"}},"official":{"repos":["opengvlab/ask-anything"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","slug":"llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","arxiv_id":"2311.17043","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.17043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17043"}},"official":{"repos":["dvlab-research/llama-vid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vamos-versatile-action-models-for-video","slug":"vamos-versatile-action-models-for-video","title":"Vamos: Versatile Action Models for Video Understanding","date":"2023-11-22","arxiv_id":"2311.13627","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vamos-versatile-action-models-for-video#ran","syntology_url":"https://syntology.ai/paper/2311.13627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13627"}},"official":{"repos":["brown-palm/Vamos"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","repositories_listed":6,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}},"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/videocon-robust-video-language-alignment-via","slug":"videocon-robust-video-language-alignment-via","title":"VideoCon: Robust Video-Language Alignment via Contrast Captions","date":"2023-11-15","arxiv_id":"2311.10111","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videocon-robust-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2311.10111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10111"}},"official":{"repos":["hritikbansal/videocon"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/testa-temporal-spatial-token-aggregation-for","slug":"testa-temporal-spatial-token-aggregation-for","title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","date":"2023-10-29","arxiv_id":"2310.19060","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/testa-temporal-spatial-token-aggregation-for#ran","syntology_url":"https://syntology.ai/paper/2310.19060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19060"}},"official":{"repos":["renshuhuai-andy/testa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-audio-visual-joint","slug":"fine-grained-audio-visual-joint","title":"Fine-grained Audio-Visual Joint Representations for Multimodal Large Language Models","date":"2023-10-09","arxiv_id":"2310.05863","repositories_listed":2,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":1,"n_instrument":7,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fine-grained-audio-visual-joint#ran","syntology_url":"https://syntology.ai/paper/2310.05863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05863"}},"official":{"repos":["briansidp/audiovisualllm","the-anonymous-bs/favor"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/one-for-all-video-conversation-is-feasible","slug":"one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","arxiv_id":"2309.15785","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/one-for-all-video-conversation-is-feasible#ran","syntology_url":"https://syntology.ai/paper/2309.15785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15785"}},"official":{"repos":["farewellthree/BT-Adapter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-i-trust-your-answer-visually-grounded","slug":"can-i-trust-your-answer-visually-grounded","title":"Can I Trust Your Answer? Visually Grounded Video Question Answering","date":"2023-09-04","arxiv_id":"2309.01327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-i-trust-your-answer-visually-grounded#ran","syntology_url":"https://syntology.ai/paper/2309.01327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01327"}},"official":{"repos":["doc-doc/next-gqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/open-vocabulary-video-question-answering-a","slug":"open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","arxiv_id":"2308.09363","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-vocabulary-video-question-answering-a#ran","syntology_url":"https://syntology.ai/paper/2308.09363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09363"}},"official":{"repos":["mlvlab/ovqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egoschema-a-diagnostic-benchmark-for-very-1#ran","syntology_url":"https://syntology.ai/paper/2308.09126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09126"}},"official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-spatio-temporal-rationales-for","slug":"discovering-spatio-temporal-rationales-for","title":"Discovering Spatio-Temporal Rationales for Video Question Answering","date":"2023-07-22","arxiv_id":"2307.12058","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discovering-spatio-temporal-rationales-for#ran","syntology_url":"https://syntology.ai/paper/2307.12058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12058"}},"official":{"repos":["yl3800/transtr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/funqa-towards-surprising-video-comprehension","slug":"funqa-towards-surprising-video-comprehension","title":"FunQA: Towards Surprising Video Comprehension","date":"2023-06-26","arxiv_id":"2306.14899","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/funqa-towards-surprising-video-comprehension#ran","syntology_url":"https://syntology.ai/paper/2306.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14899"}},"official":{"repos":["jingkang50/funqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-llama-an-instruction-tuned-audio-visual","slug":"video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","arxiv_id":"2306.02858","repositories_listed":4,"syntology":{"n":25,"n_ran":18,"n_constructed":7,"n_ran_checked":8,"n_instrument":10,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":9,"phrase":"18 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 10 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/video-llama-an-instruction-tuned-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2306.02858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02858"}},"official":{"repos":["damo-nlp-sg/video-llama"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/vast-a-vision-audio-subtitle-text-omni-1","slug":"vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","arxiv_id":"2305.18500","repositories_listed":2,"syntology":{"n":42,"n_ran":35,"n_constructed":4,"n_ran_checked":29,"n_instrument":6,"n_unverified":7,"n_honours":2,"n_violates":1,"n_no_contract":26,"n_pointer_only":8,"phrase":"35 ran (of which 4 constructed an object rather than computing a result; 29 with no instrument failure: 2 honoured, 1 violated, 26 with no contract checked; 6 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/vast-a-vision-audio-subtitle-text-omni-1#ran","syntology_url":"https://syntology.ai/paper/2305.18500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18500"}},"official":{"repos":["txh-mercury/vast"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/pali-x-on-scaling-up-a-multilingual-vision","slug":"pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","arxiv_id":"2305.18565","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":4,"n_no_contract":1,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 4 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pali-x-on-scaling-up-a-multilingual-vision#ran","syntology_url":"https://syntology.ai/paper/2305.18565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18565"}},"official":null}},{"url":"/paper/paxion-patching-action-knowledge-in-video-1","slug":"paxion-patching-action-knowledge-in-video-1","title":"Paxion: Patching Action Knowledge in Video-Language Foundation Models","date":"2023-05-18","arxiv_id":"2305.10683","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/paxion-patching-action-knowledge-in-video-1#ran","syntology_url":"https://syntology.ai/paper/2305.10683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10683"}},"official":{"repos":["mikewangwzhl/paxion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-chained-image-language-model-for-video-1","slug":"self-chained-image-language-model-for-video-1","title":"Self-Chained Image-Language Model for Video Localization and Question Answering","date":"2023-05-11","arxiv_id":"2305.06988","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-chained-image-language-model-for-video-1#ran","syntology_url":"https://syntology.ai/paper/2305.06988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06988"}},"official":{"repos":["yui010206/sevila"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/anetqa-a-large-scale-benchmark-for-fine","slug":"anetqa-a-large-scale-benchmark-for-fine","title":"ANetQA: A Large-scale Benchmark for Fine-grained Compositional Reasoning over Untrimmed Videos","date":"2023-05-04","arxiv_id":"2305.02519","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anetqa-a-large-scale-benchmark-for-fine#ran","syntology_url":"https://syntology.ai/paper/2305.02519","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02519"}},"official":{"repos":["MILVLG/anetqa-code"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-situation-hyper-graphs-for-video","slug":"learning-situation-hyper-graphs-for-video","title":"Learning Situation Hyper-Graphs for Video Question Answering","date":"2023-04-18","arxiv_id":"2304.08682","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":15,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-situation-hyper-graphs-for-video#ran","syntology_url":"https://syntology.ai/paper/2304.08682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08682"}},"official":{"repos":["aurooj/shg-vqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-instruction-tuning-1","slug":"visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":35,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 8 where Syntology's instrument failed) · 35 unverified","sample_list":"/paper/visual-instruction-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2304.08485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08485"}},"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["community","listed","named_in_paper","official"]}}},{"url":"/paper/verbs-in-action-improving-verb-understanding","slug":"verbs-in-action-improving-verb-understanding","title":"Verbs in Action: Improving verb understanding in video-language models","date":"2023-04-13","arxiv_id":"2304.06708","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/verbs-in-action-improving-verb-understanding#ran","syntology_url":"https://syntology.ai/paper/2304.06708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.06708"}},"official":{"repos":["google-research/scenic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}},{"url":"/paper/unmasked-teacher-towards-training-efficient","slug":"unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unmasked-teacher-towards-training-efficient#ran","syntology_url":"https://syntology.ai/paper/2303.16058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16058"}},"official":{"repos":["opengvlab/unmasked_teacher"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-text-as-game-players-hierarchical","slug":"video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","arxiv_id":"2303.14369","repositories_listed":4,"syntology":{"n":16,"n_ran":12,"n_constructed":7,"n_ran_checked":11,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/video-text-as-game-players-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2303.14369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14369"}},"official":{"repos":["jpthu17/HBI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/meltr-meta-loss-transformer-for-learning-to","slug":"meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","arxiv_id":"2303.13009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/meltr-meta-loss-transformer-for-learning-to#ran","syntology_url":"https://syntology.ai/paper/2303.13009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13009"}},"official":{"repos":["mlvlab/MELTR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-video-question-answering-via","slug":"contrastive-video-question-answering-via","title":"Contrastive Video Question Answering via Video Graph Transformer","date":"2023-02-27","arxiv_id":"2302.13668","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contrastive-video-question-answering-via#ran","syntology_url":"https://syntology.ai/paper/2302.13668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.13668"}},"official":{"repos":["doc-doc/covgt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/connecting-vision-and-language-with-video","slug":"connecting-vision-and-language-with-video","title":"Connecting Vision and Language with Video Localized Narratives","date":"2023-02-22","arxiv_id":"2302.11217","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/connecting-vision-and-language-with-video#ran","syntology_url":"https://syntology.ai/paper/2302.11217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11217"}},"official":{"repos":["google/video-localized-narratives"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","slug":"mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-2-a-modularized-multi-modal-foundation#ran","syntology_url":"https://syntology.ai/paper/2302.00402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00402"}},"official":{"repos":["alibaba/AliceMind"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/internvideo-general-video-foundation-models","slug":"internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","arxiv_id":"2212.03191","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internvideo-general-video-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2212.03191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03191"}},"official":{"repos":["opengvlab/internvideo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","slug":"x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-2-vlm-all-in-one-pre-trained-model-for#ran","syntology_url":"https://syntology.ai/paper/2211.12402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12402"}},"official":{"repos":["zengyan-97/x2-vlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/expectation-maximization-contrastive-learning","slug":"expectation-maximization-contrastive-learning","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations","date":"2022-11-21","arxiv_id":"2211.11427","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expectation-maximization-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2211.11427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11427"}},"official":{"repos":["jpthu17/emcl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cripp-vqa-counterfactual-reasoning-about","slug":"cripp-vqa-counterfactual-reasoning-about","title":"CRIPP-VQA: Counterfactual Reasoning about Implicit Physical Properties via Video Question Answering","date":"2022-11-07","arxiv_id":"2211.03779","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cripp-vqa-counterfactual-reasoning-about#ran","syntology_url":"https://syntology.ai/paper/2211.03779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03779"}},"official":{"repos":["maitreyapatel/cripp-vqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/long-form-video-language-pre-training-with","slug":"long-form-video-language-pre-training-with","title":"Long-Form Video-Language Pre-Training with Multimodal Temporal Contrastive Learning","date":"2022-10-12","arxiv_id":"2210.06031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/long-form-video-language-pre-training-with#ran","syntology_url":"https://syntology.ai/paper/2210.06031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06031"}},"official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-graph-transformer-for-video-question","slug":"video-graph-transformer-for-video-question","title":"Video Graph Transformer for Video Question Answering","date":"2022-07-12","arxiv_id":"2207.05342","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":5,"n_ran_checked":8,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/video-graph-transformer-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2207.05342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05342"}},"official":{"repos":["sail-sg/vgt"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}}],"record_sha256":"bcd3c579e2651f453df6a6908b763a0542b673cc3f4ee0bcb7b83d09b6bcf0f1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}