{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/ran/1","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":3,"rows_per_page":100,"rows":[1,100],"of":218,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding/papers/ran/1","prev":null,"next":"/task/video-understanding/papers/ran/2","papers":[{"url":"/paper/mcam-multimodal-causal-analysis-model-for-ego","slug":"mcam-multimodal-causal-analysis-model-for-ego","title":"MCAM: Multimodal Causal Analysis Model for Ego-Vehicle-Level Driving Video Understanding","date":"2025-07-08","arxiv_id":"2507.06072","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mcam-multimodal-causal-analysis-model-for-ego#ran","syntology_url":"https://syntology.ai/paper/2507.06072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06072"}},"official":{"repos":["sixcorepeach/mcam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/omni-video-democratizing-unified-video","slug":"omni-video-democratizing-unified-video","title":"Omni-Video: Democratizing Unified Video Understanding and Generation","date":"2025-07-08","arxiv_id":"2507.06119","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":18,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/omni-video-democratizing-unified-video#ran","syntology_url":"https://syntology.ai/paper/2507.06119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06119"}},"official":{"repos":["sais-fuxi/omni-video"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/kwai-keye-vl-technical-report","slug":"kwai-keye-vl-technical-report","title":"Kwai Keye-VL Technical Report","date":"2025-07-02","arxiv_id":"2507.01949","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kwai-keye-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.01949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01949"}},"official":{"repos":["kwai-keye/keye"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/glm-4-1v-thinking-towards-versatile","slug":"glm-4-1v-thinking-towards-versatile","title":"GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","date":"2025-07-01","arxiv_id":"2507.01006","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glm-4-1v-thinking-towards-versatile#ran","syntology_url":"https://syntology.ai/paper/2507.01006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01006"}},"official":{"repos":["thudm/glm-4.1v-thinking"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flash-vstream-efficient-real-time","slug":"flash-vstream-efficient-real-time","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","date":"2025-06-30","arxiv_id":"2506.23825","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":0,"n_instrument":7,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flash-vstream-efficient-real-time#ran","syntology_url":"https://syntology.ai/paper/2506.23825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.23825"}},"official":{"repos":["IVGSZ/Flash-VStream"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-scissor-token-compression-with-semantic","slug":"llava-scissor-token-compression-with-semantic","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","date":"2025-06-27","arxiv_id":"2506.21862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llava-scissor-token-compression-with-semantic#ran","syntology_url":"https://syntology.ai/paper/2506.21862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21862"}},"official":{"repos":["HumanMLLM/LLaVA-Scissor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-learning-of-echocardiographic","slug":"self-supervised-learning-of-echocardiographic","title":"Self-supervised Learning of Echocardiographic Video Representations via Online Cluster Distillation","date":"2025-06-13","arxiv_id":"2506.11777","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":3,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/self-supervised-learning-of-echocardiographic#ran","syntology_url":"https://syntology.ai/paper/2506.11777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11777"}},"official":{"repos":["mdivyanshu97/discovr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hopadiff-holistic-partial-aware-fourier","slug":"hopadiff-holistic-partial-aware-fourier","title":"HopaDIFF: Holistic-Partial Aware Fourier Conditioned Diffusion for Referring Human Action Segmentation in Multi-Person Scenarios","date":"2025-06-11","arxiv_id":"2506.09650","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hopadiff-holistic-partial-aware-fourier#ran","syntology_url":"https://syntology.ai/paper/2506.09650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09650"}},"official":{"repos":["kpeng9510/hopadiff"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-tuning-for-videollms","slug":"reinforcement-learning-tuning-for-videollms","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","date":"2025-06-02","arxiv_id":"2506.01908","repositories_listed":1,"syntology":{"n":17,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-tuning-for-videollms#ran","syntology_url":"https://syntology.ai/paper/2506.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01908"}},"official":{"repos":["appletea233/temporal-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/distime-distribution-based-time","slug":"distime-distribution-based-time","title":"DisTime: Distribution-based Time Representation for Video Large Language Models","date":"2025-05-30","arxiv_id":"2505.24329","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distime-distribution-based-time#ran","syntology_url":"https://syntology.ai/paper/2505.24329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24329"}},"official":{"repos":["josephzpng/distime"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prefm-online-audio-visual-event-parsing-via","slug":"prefm-online-audio-visual-event-parsing-via","title":"PreFM: Online Audio-Visual Event Parsing via Predictive Future Modeling","date":"2025-05-29","arxiv_id":"2505.23155","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prefm-online-audio-visual-event-parsing-via#ran","syntology_url":"https://syntology.ai/paper/2505.23155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23155"}},"official":{"repos":["xiaoyu-1123/prefm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videoreasonbench-can-mllms-perform-vision","slug":"videoreasonbench-can-mllms-perform-vision","title":"VideoReasonBench: Can MLLMs Perform Vision-Centric Complex Video Reasoning?","date":"2025-05-29","arxiv_id":"2505.23359","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videoreasonbench-can-mllms-perform-vision#ran","syntology_url":"https://syntology.ai/paper/2505.23359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23359"}},"official":{"repos":["llyx97/video_reason_bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/videorepa-learning-physics-for-video","slug":"videorepa-learning-physics-for-video","title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","date":"2025-05-29","arxiv_id":"2505.23656","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorepa-learning-physics-for-video#ran","syntology_url":"https://syntology.ai/paper/2505.23656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23656"}},"official":{"repos":["aHapBean/VideoREPA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vidtext-towards-comprehensive-evaluation-for","slug":"vidtext-towards-comprehensive-evaluation-for","title":"VidText: Towards Comprehensive Evaluation for Video Text Understanding","date":"2025-05-28","arxiv_id":"2505.22810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vidtext-towards-comprehensive-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.22810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22810"}},"official":{"repos":["shuyansy/vidtext"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-video-discovery-agentic-search-with-tool","slug":"deep-video-discovery-agentic-search-with-tool","title":"Deep Video Discovery: Agentic Search with Tool Use for Long-form Video Understanding","date":"2025-05-23","arxiv_id":"2505.18079","repositories_listed":0,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-video-discovery-agentic-search-with-tool#ran","syntology_url":"https://syntology.ai/paper/2505.18079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18079"}},"official":null}},{"url":"/paper/video-compression-commander-plug-and-play","slug":"video-compression-commander-plug-and-play","title":"Video Compression Commander: Plug-and-Play Inference Acceleration for Video Large Language Models","date":"2025-05-20","arxiv_id":"2505.14454","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":3,"n_instrument":6,"n_unverified":1,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/video-compression-commander-plug-and-play#ran","syntology_url":"https://syntology.ai/paper/2505.14454","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14454"}},"official":{"repos":["xuyang-liu16/vidcom2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/videoeval-pro-robust-and-realistic-long-video","slug":"videoeval-pro-robust-and-realistic-long-video","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","date":"2025-05-20","arxiv_id":"2505.14640","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videoeval-pro-robust-and-realistic-long-video#ran","syntology_url":"https://syntology.ai/paper/2505.14640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14640"}},"official":null}},{"url":"/paper/timechat-online-80-visual-tokens-are","slug":"timechat-online-80-visual-tokens-are","title":"TimeChat-Online: 80% Visual Tokens are Naturally Redundant in Streaming Videos","date":"2025-04-24","arxiv_id":"2504.17343","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/timechat-online-80-visual-tokens-are#ran","syntology_url":"https://syntology.ai/paper/2504.17343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.17343"}},"official":{"repos":["yaolinli/TimeChat-Online"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"url":"/paper/eagle-2-5-boosting-long-context-post-training","slug":"eagle-2-5-boosting-long-context-post-training","title":"Eagle 2.5: Boosting Long-Context Post-Training for Frontier Vision-Language Models","date":"2025-04-21","arxiv_id":"2504.15271","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eagle-2-5-boosting-long-context-post-training#ran","syntology_url":"https://syntology.ai/paper/2504.15271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15271"}},"official":null}},{"url":"/paper/perception-encoder-the-best-visual-embeddings","slug":"perception-encoder-the-best-visual-embeddings","title":"Perception Encoder: The best visual embeddings are not at the output of the network","date":"2025-04-17","arxiv_id":"2504.13181","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/perception-encoder-the-best-visual-embeddings#ran","syntology_url":"https://syntology.ai/paper/2504.13181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13181"}},"official":null}},{"url":"/paper/f-3-set-towards-analyzing-fast-frequent-and","slug":"f-3-set-towards-analyzing-fast-frequent-and","title":"F$^3$Set: Towards Analyzing Fast, Frequent, and Fine-grained Events from Videos","date":"2025-04-11","arxiv_id":"2504.08222","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/f-3-set-towards-analyzing-fast-frequent-and#ran","syntology_url":"https://syntology.ai/paper/2504.08222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.08222"}},"official":{"repos":["f3set/f3set"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/videochat-r1-enhancing-spatio-temporal","slug":"videochat-r1-enhancing-spatio-temporal","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","date":"2025-04-09","arxiv_id":"2504.06958","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videochat-r1-enhancing-spatio-temporal#ran","syntology_url":"https://syntology.ai/paper/2504.06958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.06958"}},"official":null}},{"url":"/paper/re-thinking-temporal-search-for-long-form","slug":"re-thinking-temporal-search-for-long-form","title":"Re-thinking Temporal Search for Long-Form Video Understanding","date":"2025-04-03","arxiv_id":"2504.02259","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/re-thinking-temporal-search-for-long-form#ran","syntology_url":"https://syntology.ai/paper/2504.02259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02259"}},"official":{"repos":["longvideohaystack/tstar"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-video-language-models-to-10k-frames","slug":"scaling-video-language-models-to-10k-frames","title":"Scaling Video-Language Models to 10K Frames via Hierarchical Differential Distillation","date":"2025-04-03","arxiv_id":"2504.02438","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scaling-video-language-models-to-10k-frames#ran","syntology_url":"https://syntology.ai/paper/2504.02438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02438"}},"official":{"repos":["steven-ccq/vilamp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/slow-fast-architecture-for-video-multi-modal","slug":"slow-fast-architecture-for-video-multi-modal","title":"Slow-Fast Architecture for Video Multi-Modal Large Language Models","date":"2025-04-02","arxiv_id":"2504.01328","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/slow-fast-architecture-for-video-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2504.01328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01328"}},"official":{"repos":["shi-labs/slow-fast-video-multimodal-llm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/spatial-r1-enhancing-mllms-in-video-spatial","slug":"spatial-r1-enhancing-mllms-in-video-spatial","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","date":"2025-04-02","arxiv_id":"2504.01805","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatial-r1-enhancing-mllms-in-video-spatial#ran","syntology_url":"https://syntology.ai/paper/2504.01805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01805"}},"official":{"repos":["ouyangkun10/spacer","ouyangkun10/spatial-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-effect-of-reinforcement","slug":"exploring-the-effect-of-reinforcement","title":"Exploring the Effect of Reinforcement Learning on Video Understanding: Insights from SEED-Bench-R1","date":"2025-03-31","arxiv_id":"2503.24376","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-the-effect-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.24376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24376"}},"official":{"repos":["tencentarc/seed-bench-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bolt-boost-large-vision-language-model","slug":"bolt-boost-large-vision-language-model","title":"BOLT: Boost Large Vision-Language Model Without Training for Long-form Video Understanding","date":"2025-03-27","arxiv_id":"2503.21483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bolt-boost-large-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2503.21483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21483"}},"official":{"repos":["sming256/bolt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentic-keyframe-search-for-video-question","slug":"agentic-keyframe-search-for-video-question","title":"Agentic Keyframe Search for Video Question Answering","date":"2025-03-20","arxiv_id":"2503.16032","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-keyframe-search-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2503.16032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16032"}},"official":{"repos":["fansunqi/akeys"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-level-instruction-injection-for-video","slug":"hybrid-level-instruction-injection-for-video","title":"Hybrid-Level Instruction Injection for Video Token Compression in Multi-modal Large Language Models","date":"2025-03-20","arxiv_id":"2503.16036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-level-instruction-injection-for-video#ran","syntology_url":"https://syntology.ai/paper/2503.16036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16036"}},"official":{"repos":["lntzm/hicom"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vispeak-visual-instruction-feedback-in","slug":"vispeak-visual-instruction-feedback-in","title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","date":"2025-03-17","arxiv_id":"2503.12769","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vispeak-visual-instruction-feedback-in#ran","syntology_url":"https://syntology.ai/paper/2503.12769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12769"}},"official":null}},{"url":"/paper/videomind-a-chain-of-lora-agent-for-long","slug":"videomind-a-chain-of-lora-agent-for-long","title":"VideoMind: A Chain-of-LoRA Agent for Long Video Reasoning","date":"2025-03-17","arxiv_id":"2503.13444","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomind-a-chain-of-lora-agent-for-long#ran","syntology_url":"https://syntology.ai/paper/2503.13444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13444"}},"official":{"repos":["yeliudev/VideoMind"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-your-vision-language-model-get-lost-in","slug":"does-your-vision-language-model-get-lost-in","title":"Does Your Vision-Language Model Get Lost in the Long Video Sampling Dilemma?","date":"2025-03-16","arxiv_id":"2503.12496","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-your-vision-language-model-get-lost-in#ran","syntology_url":"https://syntology.ai/paper/2503.12496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12496"}},"official":{"repos":["dvlab-research/LSDBench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaretake-adaptive-redundancy-reduction-to","slug":"adaretake-adaptive-redundancy-reduction-to","title":"AdaReTaKe: Adaptive Redundancy Reduction to Perceive Longer for Video-language Understanding","date":"2025-03-16","arxiv_id":"2503.12559","repositories_listed":1,"syntology":{"n":23,"n_ran":17,"n_constructed":0,"n_ran_checked":15,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/adaretake-adaptive-redundancy-reduction-to#ran","syntology_url":"https://syntology.ai/paper/2503.12559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12559"}},"official":{"repos":["sczwangxiao/video-flexreduc"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/lvagent-long-video-understanding-by-multi","slug":"lvagent-long-video-understanding-by-multi","title":"LVAgent: Long Video Understanding by Multi-Round Dynamical Collaboration of MLLM Agents","date":"2025-03-13","arxiv_id":"2503.10200","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvagent-long-video-understanding-by-multi#ran","syntology_url":"https://syntology.ai/paper/2503.10200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10200"}},"official":null}},{"url":"/paper/keyframe-oriented-vision-token-pruning","slug":"keyframe-oriented-vision-token-pruning","title":"Keyframe-oriented Vision Token Pruning: Enhancing Efficiency of Large Vision Language Models on Long-Form Video Processing","date":"2025-03-13","arxiv_id":"2503.10742","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/keyframe-oriented-vision-token-pruning#ran","syntology_url":"https://syntology.ai/paper/2503.10742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10742"}},"official":{"repos":["1999Lyd/KVTP"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vlog-video-language-models-by-generative","slug":"vlog-video-language-models-by-generative","title":"VLog: Video-Language Models by Generative Retrieval of Narration Vocabulary","date":"2025-03-12","arxiv_id":"2503.09402","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vlog-video-language-models-by-generative#ran","syntology_url":"https://syntology.ai/paper/2503.09402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09402"}},"official":{"repos":["showlab/vlog"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/quota-query-oriented-token-assignment-via-cot","slug":"quota-query-oriented-token-assignment-via-cot","title":"QuoTA: Query-oriented Token Assignment via CoT Query Decouple for Long Video Comprehension","date":"2025-03-11","arxiv_id":"2503.08689","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quota-query-oriented-token-assignment-via-cot#ran","syntology_url":"https://syntology.ai/paper/2503.08689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08689"}},"official":{"repos":["mac-automl/quota"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/egolife-towards-egocentric-life-assistant","slug":"egolife-towards-egocentric-life-assistant","title":"EgoLife: Towards Egocentric Life Assistant","date":"2025-03-05","arxiv_id":"2503.03803","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":9,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/egolife-towards-egocentric-life-assistant#ran","syntology_url":"https://syntology.ai/paper/2503.03803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03803"}},"official":{"repos":["evolvinglmms-lab/egolife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-fine-grained-hand-object-dynamics","slug":"modeling-fine-grained-hand-object-dynamics","title":"Modeling Fine-Grained Hand-Object Dynamics for Egocentric Video Representation Learning","date":"2025-03-02","arxiv_id":"2503.00986","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modeling-fine-grained-hand-object-dynamics#ran","syntology_url":"https://syntology.ai/paper/2503.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00986"}},"official":{"repos":["openrobotlab/egohod"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vrope-rotary-position-embedding-for-video","slug":"vrope-rotary-position-embedding-for-video","title":"VRoPE: Rotary Position Embedding for Video Large Language Models","date":"2025-02-17","arxiv_id":"2502.11664","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vrope-rotary-position-embedding-for-video#ran","syntology_url":"https://syntology.ai/paper/2502.11664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11664"}},"official":{"repos":["johncaged/vrope"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/svbench-a-benchmark-with-temporal-multi-turn","slug":"svbench-a-benchmark-with-temporal-multi-turn","title":"SVBench: A Benchmark with Temporal Multi-Turn Dialogues for Streaming Video Understanding","date":"2025-02-15","arxiv_id":"2502.10810","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/svbench-a-benchmark-with-temporal-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2502.10810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10810"}},"official":{"repos":["yzy-bupt/SVBench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/videorope-what-makes-for-good-video-rotary","slug":"videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","arxiv_id":"2502.05173","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorope-what-makes-for-good-video-rotary#ran","syntology_url":"https://syntology.ai/paper/2502.05173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05173"}},"official":{"repos":["wiselnn570/videorope"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/infty-video-a-training-free-approach-to-long","slug":"infty-video-a-training-free-approach-to-long","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","date":"2025-01-31","arxiv_id":"2501.19098","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infty-video-a-training-free-approach-to-long#ran","syntology_url":"https://syntology.ai/paper/2501.19098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19098"}},"official":{"repos":["deep-spin/infinite-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-video-understanding-and-multi-round","slug":"streaming-video-understanding-and-multi-round","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","date":"2025-01-23","arxiv_id":"2501.13468","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/streaming-video-understanding-and-multi-round#ran","syntology_url":"https://syntology.ai/paper/2501.13468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13468"}},"official":{"repos":["hmxiong/streamchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-3-frontier-multimodal-foundation","slug":"videollama-3-frontier-multimodal-foundation","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","date":"2025-01-22","arxiv_id":"2501.13106","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/videollama-3-frontier-multimodal-foundation#ran","syntology_url":"https://syntology.ai/paper/2501.13106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13106"}},"official":{"repos":["damo-nlp-sg/videollama3"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/valley2-exploring-multimodal-models-with","slug":"valley2-exploring-multimodal-models-with","title":"Valley2: Exploring Multimodal Models with Scalable Vision-Language Design","date":"2025-01-10","arxiv_id":"2501.05901","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/valley2-exploring-multimodal-models-with#ran","syntology_url":"https://syntology.ai/paper/2501.05901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05901"}},"official":{"repos":["bytedance/valley"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ovo-bench-how-far-is-your-video-llms-from","slug":"ovo-bench-how-far-is-your-video-llms-from","title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","date":"2025-01-09","arxiv_id":"2501.05510","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ovo-bench-how-far-is-your-video-llms-from#ran","syntology_url":"https://syntology.ai/paper/2501.05510","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05510"}},"official":{"repos":["joeleelyf/ovo-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-specialized-visual-encoders-for","slug":"unifying-specialized-visual-encoders-for","title":"Unifying Specialized Visual Encoders for Video Language Models","date":"2025-01-02","arxiv_id":"2501.01426","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-specialized-visual-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2501.01426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01426"}},"official":{"repos":["princetonvisualai/merv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/framefusion-combining-similarity-and","slug":"framefusion-combining-similarity-and","title":"FrameFusion: Combining Similarity and Importance for Video Token Reduction on Large Visual Language Models","date":"2024-12-30","arxiv_id":"2501.01986","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/framefusion-combining-similarity-and#ran","syntology_url":"https://syntology.ai/paper/2501.01986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01986"}},"official":{"repos":["thu-nics/framefusion"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/retake-reducing-temporal-and-knowledge","slug":"retake-reducing-temporal-and-knowledge","title":"ReTaKe: Reducing Temporal and Knowledge Redundancy for Long Video Understanding","date":"2024-12-29","arxiv_id":"2412.20504","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/retake-reducing-temporal-and-knowledge#ran","syntology_url":"https://syntology.ai/paper/2412.20504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.20504"}},"official":{"repos":["sczwangxiao/video-retake"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/flashvtg-feature-layering-and-adaptive-score","slug":"flashvtg-feature-layering-and-adaptive-score","title":"FlashVTG: Feature Layering and Adaptive Score Handling Network for Video Temporal Grounding","date":"2024-12-18","arxiv_id":"2412.13441","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flashvtg-feature-layering-and-adaptive-score#ran","syntology_url":"https://syntology.ai/paper/2412.13441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13441"}},"official":{"repos":["zhuo-cao/flashvtg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/instructseg-unifying-instructed-visual","slug":"instructseg-unifying-instructed-visual","title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","date":"2024-12-18","arxiv_id":"2412.14006","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/instructseg-unifying-instructed-visual#ran","syntology_url":"https://syntology.ai/paper/2412.14006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14006"}},"official":{"repos":["congvvc/instructseg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/linvt-empower-your-image-level-large-language","slug":"linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","arxiv_id":"2412.05185","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":6,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/linvt-empower-your-image-level-large-language#ran","syntology_url":"https://syntology.ai/paper/2412.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05185"}},"official":{"repos":["gls0425/linvt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/visionzip-longer-is-better-but-not-necessary","slug":"visionzip-longer-is-better-but-not-necessary","title":"VisionZip: Longer is Better but Not Necessary in Vision Language Models","date":"2024-12-05","arxiv_id":"2412.04467","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visionzip-longer-is-better-but-not-necessary#ran","syntology_url":"https://syntology.ai/paper/2412.04467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04467"}},"official":{"repos":["dvlab-research/visionzip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/aim-adaptive-inference-of-multi-modal-llms","slug":"aim-adaptive-inference-of-multi-modal-llms","title":"AIM: Adaptive Inference of Multi-Modal LLMs via Token Merging and Pruning","date":"2024-12-04","arxiv_id":"2412.03248","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/aim-adaptive-inference-of-multi-modal-llms#ran","syntology_url":"https://syntology.ai/paper/2412.03248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03248"}},"official":{"repos":["lavi-lab/aim"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/inst-it-boosting-multimodal-instance","slug":"inst-it-boosting-multimodal-instance","title":"Inst-IT: Boosting Multimodal Instance Understanding via Explicit Visual Prompt Instruction Tuning","date":"2024-12-04","arxiv_id":"2412.03565","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inst-it-boosting-multimodal-instance#ran","syntology_url":"https://syntology.ai/paper/2412.03565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03565"}},"official":{"repos":["inst-it/inst-it"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videoicl-confidence-based-iterative-in","slug":"videoicl-confidence-based-iterative-in","title":"VideoICL: Confidence-based Iterative In-context Learning for Out-of-Distribution Video Understanding","date":"2024-12-03","arxiv_id":"2412.02186","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videoicl-confidence-based-iterative-in#ran","syntology_url":"https://syntology.ai/paper/2412.02186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02186"}},"official":{"repos":["kangsankim07/videoicl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/longvale-vision-audio-language-event","slug":"longvale-vision-audio-language-event","title":"LongVALE: Vision-Audio-Language-Event Benchmark Towards Time-Aware Omni-Modal Perception of Long Videos","date":"2024-11-29","arxiv_id":"2411.19772","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/longvale-vision-audio-language-event#ran","syntology_url":"https://syntology.ai/paper/2411.19772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19772"}},"official":{"repos":["ttgeng233/LongVALE"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/video-rag-visually-aligned-retrieval","slug":"video-rag-visually-aligned-retrieval","title":"Video-RAG: Visually-aligned Retrieval-Augmented Long Video Comprehension","date":"2024-11-20","arxiv_id":"2411.13093","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-rag-visually-aligned-retrieval#ran","syntology_url":"https://syntology.ai/paper/2411.13093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13093"}},"official":{"repos":["leon1207/video-rag-master"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/teaching-vlms-to-localize-specific-objects","slug":"teaching-vlms-to-localize-specific-objects","title":"Teaching VLMs to Localize Specific Objects from In-context Examples","date":"2024-11-20","arxiv_id":"2411.13317","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-vlms-to-localize-specific-objects#ran","syntology_url":"https://syntology.ai/paper/2411.13317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13317"}},"official":{"repos":["sivandoveh/iploc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ts-llava-constructing-visual-tokens-through","slug":"ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","arxiv_id":"2411.11066","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ts-llava-constructing-visual-tokens-through#ran","syntology_url":"https://syntology.ai/paper/2411.11066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11066"}},"official":{"repos":["tingyu215/ts-llava"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ppllava-varied-video-sequence-understanding","slug":"ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ppllava-varied-video-sequence-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.02327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02327"}},"official":{"repos":["farewellthree/ppllava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/tomato-assessing-visual-temporal-reasoning","slug":"tomato-assessing-visual-temporal-reasoning","title":"TOMATO: Assessing Visual Temporal Reasoning Capabilities in Multimodal Foundation Models","date":"2024-10-30","arxiv_id":"2410.23266","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tomato-assessing-visual-temporal-reasoning#ran","syntology_url":"https://syntology.ai/paper/2410.23266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23266"}},"official":{"repos":["yale-nlp/TOMATO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timesuite-improving-mllms-for-long-video","slug":"timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","arxiv_id":"2410.19702","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timesuite-improving-mllms-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2410.19702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19702"}},"official":null}},{"url":"/paper/longvu-spatiotemporal-adaptive-compression","slug":"longvu-spatiotemporal-adaptive-compression","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","date":"2024-10-22","arxiv_id":"2410.17434","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvu-spatiotemporal-adaptive-compression#ran","syntology_url":"https://syntology.ai/paper/2410.17434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17434"}},"official":{"repos":["Vision-CAIR/LongVU"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videgothink-assessing-egocentric-video","slug":"videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","arxiv_id":"2410.11623","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":5,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videgothink-assessing-egocentric-video#ran","syntology_url":"https://syntology.ai/paper/2410.11623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11623"}},"official":null}},{"url":"/paper/temporalbench-benchmarking-fine-grained","slug":"temporalbench-benchmarking-fine-grained","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","date":"2024-10-14","arxiv_id":"2410.10818","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporalbench-benchmarking-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.10818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10818"}},"official":{"repos":["mu-cai/TemporalBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trace-temporal-grounding-video-llm-via-causal","slug":"trace-temporal-grounding-video-llm-via-causal","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","date":"2024-10-08","arxiv_id":"2410.05643","repositories_listed":1,"syntology":{"n":16,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/trace-temporal-grounding-video-llm-via-causal#ran","syntology_url":"https://syntology.ai/paper/2410.05643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05643"}},"official":{"repos":["gyxxyg/trace"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-temporal-modeling-of-video-llms-via","slug":"enhancing-temporal-modeling-of-video-llms-via","title":"Enhancing Temporal Modeling of Video LLMs via Time Gating","date":"2024-10-08","arxiv_id":"2410.05714","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/enhancing-temporal-modeling-of-video-llms-via#ran","syntology_url":"https://syntology.ai/paper/2410.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05714"}},"official":{"repos":["lavi-lab/tg-vid"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sparsevlm-visual-token-sparsification-for","slug":"sparsevlm-visual-token-sparsification-for","title":"SparseVLM: Visual Token Sparsification for Efficient Vision-Language Model Inference","date":"2024-10-06","arxiv_id":"2410.04417","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sparsevlm-visual-token-sparsification-for#ran","syntology_url":"https://syntology.ai/paper/2410.04417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04417"}},"official":{"repos":["gumpest/sparsevlms"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-videollm-sharpening-fine-grained","slug":"grounded-videollm-sharpening-fine-grained","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","date":"2024-10-04","arxiv_id":"2410.03290","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-videollm-sharpening-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.03290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03290"}},"official":{"repos":["whb139426/grounded-video-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-xl-extra-long-vision-language-model-for","slug":"video-xl-extra-long-vision-language-model-for","title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","date":"2024-09-22","arxiv_id":"2409.14485","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-xl-extra-long-vision-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2409.14485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14485"}},"official":{"repos":["vectorspacelab/video-xl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/longllava-scaling-multi-modal-llms-to-1000","slug":"longllava-scaling-multi-modal-llms-to-1000","title":"LongLLaVA: Scaling Multi-modal LLMs to 1000 Images Efficiently via a Hybrid Architecture","date":"2024-09-04","arxiv_id":"2409.02889","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longllava-scaling-multi-modal-llms-to-1000#ran","syntology_url":"https://syntology.ai/paper/2409.02889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02889"}},"official":{"repos":["freedomintelligence/longllava"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cogvlm2-visual-language-models-for-image-and","slug":"cogvlm2-visual-language-models-for-image-and","title":"CogVLM2: Visual Language Models for Image and Video Understanding","date":"2024-08-29","arxiv_id":"2408.16500","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cogvlm2-visual-language-models-for-image-and#ran","syntology_url":"https://syntology.ai/paper/2408.16500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16500"}},"official":{"repos":["thudm/cogvlm2","thudm/glm-4"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/video-ccam-enhancing-video-language","slug":"video-ccam-enhancing-video-language","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","date":"2024-08-26","arxiv_id":"2408.14023","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-ccam-enhancing-video-language#ran","syntology_url":"https://syntology.ai/paper/2408.14023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14023"}},"official":{"repos":["qq-mm/video-ccam"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hat-history-augmented-anchor-transformer-for","slug":"hat-history-augmented-anchor-transformer-for","title":"HAT: History-Augmented Anchor Transformer for Online Temporal Action Localization","date":"2024-08-12","arxiv_id":"2408.06437","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":3,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 3 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hat-history-augmented-anchor-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2408.06437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06437"}},"official":{"repos":["sakibreza/eccv24-hat"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":3,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egocvr-an-egocentric-benchmark-for-fine","slug":"egocvr-an-egocentric-benchmark-for-fine","title":"EgoCVR: An Egocentric Benchmark for Fine-Grained Composed Video Retrieval","date":"2024-07-23","arxiv_id":"2407.16658","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/egocvr-an-egocentric-benchmark-for-fine#ran","syntology_url":"https://syntology.ai/paper/2407.16658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16658"}},"official":{"repos":["explainableml/egocvr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/longvideobench-a-benchmark-for-long-context","slug":"longvideobench-a-benchmark-for-long-context","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","date":"2024-07-22","arxiv_id":"2407.15754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvideobench-a-benchmark-for-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.15754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15754"}},"official":{"repos":["longvideobench/longvideobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slowfast-llava-a-strong-training-free","slug":"slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","arxiv_id":"2407.15841","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slowfast-llava-a-strong-training-free#ran","syntology_url":"https://syntology.ai/paper/2407.15841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15841"}},"official":{"repos":["apple/ml-slowfast-llava"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videoeval-comprehensive-benchmark-suite-for","slug":"videoeval-comprehensive-benchmark-suite-for","title":"VideoEval: Comprehensive Benchmark Suite for Low-Cost Evaluation of Video Foundation Model","date":"2024-07-09","arxiv_id":"2407.06491","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/videoeval-comprehensive-benchmark-suite-for#ran","syntology_url":"https://syntology.ai/paper/2407.06491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06491"}},"official":{"repos":["leexinhao/VideoEval"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/video-star-self-training-enables-video","slug":"video-star-self-training-enables-video","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","date":"2024-07-08","arxiv_id":"2407.06189","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-star-self-training-enables-video#ran","syntology_url":"https://syntology.ai/paper/2407.06189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06189"}},"official":{"repos":["orrzohar/Video-STaR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/internlm-xcomposer-2-5-a-versatile-large","slug":"internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","arxiv_id":"2407.03320","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internlm-xcomposer-2-5-a-versatile-large#ran","syntology_url":"https://syntology.ai/paper/2407.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03320"}},"official":{"repos":["internlm/internlm-xcomposer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/tarsier-recipes-for-training-and-evaluating-1","slug":"tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","arxiv_id":"2407.00634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier-recipes-for-training-and-evaluating-1#ran","syntology_url":"https://syntology.ai/paper/2407.00634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00634"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omg-llava-bridging-image-level-object-level","slug":"omg-llava-bridging-image-level-object-level","title":"OMG-LLaVA: Bridging Image-level, Object-level, Pixel-level Reasoning and Understanding","date":"2024-06-27","arxiv_id":"2406.19389","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omg-llava-bridging-image-level-object-level#ran","syntology_url":"https://syntology.ai/paper/2406.19389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19389"}},"official":null}},{"url":"/paper/towards-event-oriented-long-video","slug":"towards-event-oriented-long-video","title":"Towards Event-oriented Long Video Understanding","date":"2024-06-20","arxiv_id":"2406.14129","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-event-oriented-long-video#ran","syntology_url":"https://syntology.ai/paper/2406.14129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14129"}},"official":{"repos":["rucaibox/event-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmbench-video-a-long-form-multi-shot","slug":"mmbench-video-a-long-form-multi-shot","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","date":"2024-06-20","arxiv_id":"2406.14515","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmbench-video-a-long-form-multi-shot#ran","syntology_url":"https://syntology.ai/paper/2406.14515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14515"}},"official":{"repos":["open-compass/vlmevalkit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slot-state-space-models","slug":"slot-state-space-models","title":"Slot State Space Models","date":"2024-06-18","arxiv_id":"2406.12272","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slot-state-space-models#ran","syntology_url":"https://syntology.ai/paper/2406.12272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12272"}},"official":{"repos":["jindongjiang/slotssms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/needle-in-a-video-haystack-a-scalable","slug":"needle-in-a-video-haystack-a-scalable","title":"Needle In A Video Haystack: A Scalable Synthetic Evaluator for Video MLLMs","date":"2024-06-13","arxiv_id":"2406.09367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/needle-in-a-video-haystack-a-scalable#ran","syntology_url":"https://syntology.ai/paper/2406.09367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09367"}},"official":{"repos":["joez17/videoniah"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videogpt-integrating-image-and-video-encoders#ran","syntology_url":"https://syntology.ai/paper/2406.09418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09418"}},"official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lvbench-an-extreme-long-video-understanding","slug":"lvbench-an-extreme-long-video-understanding","title":"LVBench: An Extreme Long Video Understanding Benchmark","date":"2024-06-12","arxiv_id":"2406.08035","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvbench-an-extreme-long-video-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.08035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08035"}},"official":{"repos":["THUDM/LVBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmworld-towards-multi-discipline-multi","slug":"mmworld-towards-multi-discipline-multi","title":"MMWorld: Towards Multi-discipline Multi-faceted World Model Evaluation in Videos","date":"2024-06-12","arxiv_id":"2406.08407","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmworld-towards-multi-discipline-multi#ran","syntology_url":"https://syntology.ai/paper/2406.08407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08407"}},"official":{"repos":["eric-ai-lab/mmworld"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vript-a-video-is-worth-thousands-of-words","slug":"vript-a-video-is-worth-thousands-of-words","title":"Vript: A Video Is Worth Thousands of Words","date":"2024-06-10","arxiv_id":"2406.06040","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vript-a-video-is-worth-thousands-of-words#ran","syntology_url":"https://syntology.ai/paper/2406.06040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06040"}},"official":{"repos":["mutonix/vript"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mlvu-a-comprehensive-benchmark-for-multi-task","slug":"mlvu-a-comprehensive-benchmark-for-multi-task","title":"MLVU: Benchmarking Multi-task Long Video Understanding","date":"2024-06-06","arxiv_id":"2406.04264","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mlvu-a-comprehensive-benchmark-for-multi-task#ran","syntology_url":"https://syntology.ai/paper/2406.04264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04264"}},"official":{"repos":["junjie99/mlvu","FlagOpen/FlagEmbedding"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/demamba-ai-generated-video-detection-on","slug":"demamba-ai-generated-video-detection-on","title":"DeMamba: AI-Generated Video Detection on Million-Scale GenVideo Benchmark","date":"2024-05-30","arxiv_id":"2405.19707","repositories_listed":1,"syntology":{"n":19,"n_ran":13,"n_constructed":0,"n_ran_checked":9,"n_instrument":4,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/demamba-ai-generated-video-detection-on#ran","syntology_url":"https://syntology.ai/paper/2405.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19707"}},"official":{"repos":["chenhaoxing/DeMamba"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}}],"record_sha256":"a7afbcbc335d58a187733eab14a793ad2c933febcd786db5f0e2e890833933f5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}