{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/2","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":12,"rows_per_page":100,"rows":[101,200],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding","next":"/task/video-understanding/papers/3","papers":[{"url":"/paper/mcam-multimodal-causal-analysis-model-for-ego","slug":"mcam-multimodal-causal-analysis-model-for-ego","title":"MCAM: Multimodal Causal Analysis Model for Ego-Vehicle-Level Driving Video Understanding","date":"2025-07-08","arxiv_id":"2507.06072","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mcam-multimodal-causal-analysis-model-for-ego#ran","syntology_url":"https://syntology.ai/paper/2507.06072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06072"}},"official":{"repos":["sixcorepeach/mcam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/omni-video-democratizing-unified-video","slug":"omni-video-democratizing-unified-video","title":"Omni-Video: Democratizing Unified Video Understanding and Generation","date":"2025-07-08","arxiv_id":"2507.06119","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":18,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/omni-video-democratizing-unified-video#ran","syntology_url":"https://syntology.ai/paper/2507.06119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06119"}},"official":{"repos":["sais-fuxi/omni-video"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/kwai-keye-vl-technical-report","slug":"kwai-keye-vl-technical-report","title":"Kwai Keye-VL Technical Report","date":"2025-07-02","arxiv_id":"2507.01949","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kwai-keye-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.01949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01949"}},"official":{"repos":["kwai-keye/keye"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/glm-4-1v-thinking-towards-versatile","slug":"glm-4-1v-thinking-towards-versatile","title":"GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","date":"2025-07-01","arxiv_id":"2507.01006","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glm-4-1v-thinking-towards-versatile#ran","syntology_url":"https://syntology.ai/paper/2507.01006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01006"}},"official":{"repos":["thudm/glm-4.1v-thinking"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flash-vstream-efficient-real-time","slug":"flash-vstream-efficient-real-time","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","date":"2025-06-30","arxiv_id":"2506.23825","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":0,"n_instrument":7,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flash-vstream-efficient-real-time#ran","syntology_url":"https://syntology.ai/paper/2506.23825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.23825"}},"official":{"repos":["IVGSZ/Flash-VStream"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-scissor-token-compression-with-semantic","slug":"llava-scissor-token-compression-with-semantic","title":"LLaVA-Scissor: Token Compression with Semantic Connected Components for Video LLMs","date":"2025-06-27","arxiv_id":"2506.21862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llava-scissor-token-compression-with-semantic#ran","syntology_url":"https://syntology.ai/paper/2506.21862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21862"}},"official":{"repos":["HumanMLLM/LLaVA-Scissor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/task-aware-kv-compression-for-cost-effective","slug":"task-aware-kv-compression-for-cost-effective","title":"Task-Aware KV Compression For Cost-Effective Long Video Understanding","date":"2025-06-26","arxiv_id":"2506.21184","repositories_listed":1,"syntology":null},{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/eva02-at-egocentric-video-language","slug":"eva02-at-egocentric-video-language","title":"EVA02-AT: Egocentric Video-Language Understanding with Spatial-Temporal Rotary Positional Embeddings and Symmetric Optimization","date":"2025-06-17","arxiv_id":"2506.14356","repositories_listed":1,"syntology":null},{"url":"/paper/m-3-vos-multi-phase-multi-transition-and-1","slug":"m-3-vos-multi-phase-multi-transition-and-1","title":"M^3-VOS: Multi-Phase, Multi-Transition, and Multi-Scenery Video Object Segmentation","date":"2025-06-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-of-echocardiographic","slug":"self-supervised-learning-of-echocardiographic","title":"Self-supervised Learning of Echocardiographic Video Representations via Online Cluster Distillation","date":"2025-06-13","arxiv_id":"2506.11777","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":3,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/self-supervised-learning-of-echocardiographic#ran","syntology_url":"https://syntology.ai/paper/2506.11777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11777"}},"official":{"repos":["mdivyanshu97/discovr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/videodeepresearch-long-video-understanding","slug":"videodeepresearch-long-video-understanding","title":"VideoDeepResearch: Long Video Understanding With Agentic Tool Using","date":"2025-06-12","arxiv_id":"2506.10821","repositories_listed":1,"syntology":null},{"url":"/paper/hopadiff-holistic-partial-aware-fourier","slug":"hopadiff-holistic-partial-aware-fourier","title":"HopaDIFF: Holistic-Partial Aware Fourier Conditioned Diffusion for Referring Human Action Segmentation in Multi-Person Scenarios","date":"2025-06-11","arxiv_id":"2506.09650","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hopadiff-holistic-partial-aware-fourier#ran","syntology_url":"https://syntology.ai/paper/2506.09650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09650"}},"official":{"repos":["kpeng9510/hopadiff"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cyberv-cybernetics-for-test-time-scaling-in","slug":"cyberv-cybernetics-for-test-time-scaling-in","title":"CyberV: Cybernetics for Test-time Scaling in Video Understanding","date":"2025-06-09","arxiv_id":"2506.07971","repositories_listed":1,"syntology":null},{"url":"/paper/egovlm-policy-optimization-for-egocentric","slug":"egovlm-policy-optimization-for-egocentric","title":"EgoVLM: Policy Optimization for Egocentric Video Understanding","date":"2025-06-03","arxiv_id":"2506.03097","repositories_listed":1,"syntology":null},{"url":"/paper/metok-multi-stage-event-based-token","slug":"metok-multi-stage-event-based-token","title":"METok: Multi-Stage Event-based Token Compression for Efficient Long Video Understanding","date":"2025-06-03","arxiv_id":"2506.02850","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-tuning-for-videollms","slug":"reinforcement-learning-tuning-for-videollms","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","date":"2025-06-02","arxiv_id":"2506.01908","repositories_listed":1,"syntology":{"n":17,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-tuning-for-videollms#ran","syntology_url":"https://syntology.ai/paper/2506.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01908"}},"official":{"repos":["appletea233/temporal-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/distime-distribution-based-time","slug":"distime-distribution-based-time","title":"DisTime: Distribution-based Time Representation for Video Large Language Models","date":"2025-05-30","arxiv_id":"2505.24329","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distime-distribution-based-time#ran","syntology_url":"https://syntology.ai/paper/2505.24329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24329"}},"official":{"repos":["josephzpng/distime"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/silvr-a-simple-language-based-video-reasoning","slug":"silvr-a-simple-language-based-video-reasoning","title":"SiLVR: A Simple Language-based Video Reasoning Framework","date":"2025-05-30","arxiv_id":"2505.24869","repositories_listed":1,"syntology":null},{"url":"/paper/videocad-a-large-scale-video-dataset-for","slug":"videocad-a-large-scale-video-dataset-for","title":"VideoCAD: A Large-Scale Video Dataset for Learning UI Interactions and 3D Reasoning from CAD Software","date":"2025-05-30","arxiv_id":"2505.24838","repositories_listed":1,"syntology":null},{"url":"/paper/one-trajectory-one-token-grounded-video","slug":"one-trajectory-one-token-grounded-video","title":"One Trajectory, One Token: Grounded Video Tokenization via Panoptic Sub-object Trajectory","date":"2025-05-29","arxiv_id":"2505.23617","repositories_listed":1,"syntology":null},{"url":"/paper/prefm-online-audio-visual-event-parsing-via","slug":"prefm-online-audio-visual-event-parsing-via","title":"PreFM: Online Audio-Visual Event Parsing via Predictive Future Modeling","date":"2025-05-29","arxiv_id":"2505.23155","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prefm-online-audio-visual-event-parsing-via#ran","syntology_url":"https://syntology.ai/paper/2505.23155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23155"}},"official":{"repos":["xiaoyu-1123/prefm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scalelong-a-multi-timescale-benchmark-for","slug":"scalelong-a-multi-timescale-benchmark-for","title":"ScaleLong: A Multi-Timescale Benchmark for Long Video Understanding","date":"2025-05-29","arxiv_id":"2505.23922","repositories_listed":1,"syntology":null},{"url":"/paper/videoreasonbench-can-mllms-perform-vision","slug":"videoreasonbench-can-mllms-perform-vision","title":"VideoReasonBench: Can MLLMs Perform Vision-Centric Complex Video Reasoning?","date":"2025-05-29","arxiv_id":"2505.23359","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videoreasonbench-can-mllms-perform-vision#ran","syntology_url":"https://syntology.ai/paper/2505.23359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23359"}},"official":{"repos":["llyx97/video_reason_bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/videorepa-learning-physics-for-video","slug":"videorepa-learning-physics-for-video","title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","date":"2025-05-29","arxiv_id":"2505.23656","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorepa-learning-physics-for-video#ran","syntology_url":"https://syntology.ai/paper/2505.23656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23656"}},"official":{"repos":["aHapBean/VideoREPA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vidtext-towards-comprehensive-evaluation-for","slug":"vidtext-towards-comprehensive-evaluation-for","title":"VidText: Towards Comprehensive Evaluation for Video Text Understanding","date":"2025-05-28","arxiv_id":"2505.22810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vidtext-towards-comprehensive-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.22810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22810"}},"official":{"repos":["shuyansy/vidtext"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/museg-reinforcing-video-temporal","slug":"museg-reinforcing-video-temporal","title":"MUSEG: Reinforcing Video Temporal Understanding via Timestamp-Aware Multi-Segment Grounding","date":"2025-05-27","arxiv_id":"2505.20715","repositories_listed":1,"syntology":null},{"url":"/paper/tuna-comprehensive-fine-grained-temporal","slug":"tuna-comprehensive-fine-grained-temporal","title":"TUNA: Comprehensive Fine-grained Temporal Understanding Evaluation on Dense Dynamic Videos","date":"2025-05-26","arxiv_id":"2505.20124","repositories_listed":1,"syntology":null},{"url":"/paper/fact-r1-towards-explainable-video","slug":"fact-r1-towards-explainable-video","title":"Fact-R1: Towards Explainable Video Misinformation Detection with Deep Reasoning","date":"2025-05-22","arxiv_id":"2505.16836","repositories_listed":1,"syntology":null},{"url":"/paper/quickvideo-real-time-long-video-understanding","slug":"quickvideo-real-time-long-video-understanding","title":"QuickVideo: Real-Time Long Video Understanding with System Algorithm Co-Design","date":"2025-05-22","arxiv_id":"2505.16175","repositories_listed":1,"syntology":null},{"url":"/paper/viqagent-zero-shot-video-question-answering","slug":"viqagent-zero-shot-video-question-answering","title":"ViQAgent: Zero-Shot Video Question Answering via Agent with Open-Vocabulary Grounding Validation","date":"2025-05-21","arxiv_id":"2505.15928","repositories_listed":1,"syntology":null},{"url":"/paper/a-challenge-to-build-neuro-symbolic-video","slug":"a-challenge-to-build-neuro-symbolic-video","title":"A Challenge to Build Neuro-Symbolic Video Agents","date":"2025-05-20","arxiv_id":"2505.13851","repositories_listed":1,"syntology":null},{"url":"/paper/lovr-a-benchmark-for-long-video-retrieval-in","slug":"lovr-a-benchmark-for-long-video-retrieval-in","title":"LoVR: A Benchmark for Long Video Retrieval in Multimodal Contexts","date":"2025-05-20","arxiv_id":"2505.13928","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-oriented-recipe-for-transferring","slug":"temporal-oriented-recipe-for-transferring","title":"Temporal-Oriented Recipe for Transferring Large Vision-Language Model to Video Understanding","date":"2025-05-19","arxiv_id":"2505.12605","repositories_listed":1,"syntology":null},{"url":"/paper/vcrbench-exploring-long-form-causal-reasoning","slug":"vcrbench-exploring-long-form-causal-reasoning","title":"VCRBench: Exploring Long-form Causal Reasoning Capabilities of Large Video Language Models","date":"2025-05-13","arxiv_id":"2505.08455","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-weighted-image-event-multimodal","slug":"uncertainty-weighted-image-event-multimodal","title":"Uncertainty-Weighted Image-Event Multimodal Fusion for Video Anomaly Detection","date":"2025-05-05","arxiv_id":"2505.02393","repositories_listed":1,"syntology":null},{"url":"/paper/tempura-temporal-event-masked-prediction-and","slug":"tempura-temporal-event-masked-prediction-and","title":"TEMPURA: Temporal Event Masked Prediction and Understanding for Reasoning in Action","date":"2025-05-02","arxiv_id":"2505.01583","repositories_listed":1,"syntology":null},{"url":"/paper/videohallu-evaluating-and-mitigating-multi","slug":"videohallu-evaluating-and-mitigating-multi","title":"VideoHallu: Evaluating and Mitigating Multi-modal Hallucinations on Synthetic Video Understanding","date":"2025-05-02","arxiv_id":"2505.01481","repositories_listed":1,"syntology":null},{"url":"/paper/seriesbench-a-benchmark-for-narrative-driven","slug":"seriesbench-a-benchmark-for-narrative-driven","title":"SeriesBench: A Benchmark for Narrative-Driven Drama Series Understanding","date":"2025-04-30","arxiv_id":"2504.21435","repositories_listed":1,"syntology":null},{"url":"/paper/videomultiagents-a-multi-agent-framework-for","slug":"videomultiagents-a-multi-agent-framework-for","title":"VideoMultiAgents: A Multi-Agent Framework for Video Question Answering","date":"2025-04-25","arxiv_id":"2504.20091","repositories_listed":1,"syntology":null},{"url":"/paper/eagle-2-5-boosting-long-context-post-training","slug":"eagle-2-5-boosting-long-context-post-training","title":"Eagle 2.5: Boosting Long-Context Post-Training for Frontier Vision-Language Models","date":"2025-04-21","arxiv_id":"2504.15271","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eagle-2-5-boosting-long-context-post-training#ran","syntology_url":"https://syntology.ai/paper/2504.15271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15271"}},"official":null}},{"url":"/paper/iv-bench-a-benchmark-for-image-grounded-video","slug":"iv-bench-a-benchmark-for-image-grounded-video","title":"IV-Bench: A Benchmark for Image-Grounded Video Perception and Reasoning in Multimodal LLMs","date":"2025-04-21","arxiv_id":"2504.15415","repositories_listed":1,"syntology":null},{"url":"/paper/are-vision-llms-road-ready-a-comprehensive","slug":"are-vision-llms-road-ready-a-comprehensive","title":"Are Vision LLMs Road-Ready? A Comprehensive Benchmark for Safety-Critical Driving Video Understanding","date":"2025-04-20","arxiv_id":"2504.14526","repositories_listed":1,"syntology":null},{"url":"/paper/perception-encoder-the-best-visual-embeddings","slug":"perception-encoder-the-best-visual-embeddings","title":"Perception Encoder: The best visual embeddings are not at the output of the network","date":"2025-04-17","arxiv_id":"2504.13181","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/perception-encoder-the-best-visual-embeddings#ran","syntology_url":"https://syntology.ai/paper/2504.13181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13181"}},"official":null}},{"url":"/paper/perceptionlm-open-access-data-and-models-for","slug":"perceptionlm-open-access-data-and-models-for","title":"PerceptionLM: Open-Access Data and Models for Detailed Visual Understanding","date":"2025-04-17","arxiv_id":"2504.13180","repositories_listed":1,"syntology":null},{"url":"/paper/vistadpo-video-hierarchical-spatial-temporal","slug":"vistadpo-video-hierarchical-spatial-temporal","title":"VistaDPO: Video Hierarchical Spatial-Temporal Direct Preference Optimization for Large Video Models","date":"2025-04-17","arxiv_id":"2504.13122","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-long-video-modeling-based-on","slug":"multimodal-long-video-modeling-based-on","title":"Multimodal Long Video Modeling Based on Temporal Dynamic Context","date":"2025-04-14","arxiv_id":"2504.10443","repositories_listed":1,"syntology":null},{"url":"/paper/tinyllava-video-r1-towards-smaller-lmms-for","slug":"tinyllava-video-r1-towards-smaller-lmms-for","title":"TinyLLaVA-Video-R1: Towards Smaller LMMs for Video Reasoning","date":"2025-04-13","arxiv_id":"2504.09641","repositories_listed":1,"syntology":null},{"url":"/paper/f-3-set-towards-analyzing-fast-frequent-and","slug":"f-3-set-towards-analyzing-fast-frequent-and","title":"F$^3$Set: Towards Analyzing Fast, Frequent, and Fine-grained Events from Videos","date":"2025-04-11","arxiv_id":"2504.08222","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/f-3-set-towards-analyzing-fast-frequent-and#ran","syntology_url":"https://syntology.ai/paper/2504.08222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.08222"}},"official":{"repos":["f3set/f3set"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/videochat-r1-enhancing-spatio-temporal","slug":"videochat-r1-enhancing-spatio-temporal","title":"VideoChat-R1: Enhancing Spatio-Temporal Perception via Reinforcement Fine-Tuning","date":"2025-04-09","arxiv_id":"2504.06958","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videochat-r1-enhancing-spatio-temporal#ran","syntology_url":"https://syntology.ai/paper/2504.06958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.06958"}},"official":null}},{"url":"/paper/re-thinking-temporal-search-for-long-form","slug":"re-thinking-temporal-search-for-long-form","title":"Re-thinking Temporal Search for Long-Form Video Understanding","date":"2025-04-03","arxiv_id":"2504.02259","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/re-thinking-temporal-search-for-long-form#ran","syntology_url":"https://syntology.ai/paper/2504.02259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02259"}},"official":{"repos":["longvideohaystack/tstar"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-video-language-models-to-10k-frames","slug":"scaling-video-language-models-to-10k-frames","title":"Scaling Video-Language Models to 10K Frames via Hierarchical Differential Distillation","date":"2025-04-03","arxiv_id":"2504.02438","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scaling-video-language-models-to-10k-frames#ran","syntology_url":"https://syntology.ai/paper/2504.02438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02438"}},"official":{"repos":["steven-ccq/vilamp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/slow-fast-architecture-for-video-multi-modal","slug":"slow-fast-architecture-for-video-multi-modal","title":"Slow-Fast Architecture for Video Multi-Modal Large Language Models","date":"2025-04-02","arxiv_id":"2504.01328","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/slow-fast-architecture-for-video-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2504.01328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01328"}},"official":{"repos":["shi-labs/slow-fast-video-multimodal-llm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/exploring-the-effect-of-reinforcement","slug":"exploring-the-effect-of-reinforcement","title":"Exploring the Effect of Reinforcement Learning on Video Understanding: Insights from SEED-Bench-R1","date":"2025-03-31","arxiv_id":"2503.24376","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-the-effect-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.24376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24376"}},"official":{"repos":["tencentarc/seed-bench-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bolt-boost-large-vision-language-model","slug":"bolt-boost-large-vision-language-model","title":"BOLT: Boost Large Vision-Language Model Without Training for Long-form Video Understanding","date":"2025-03-27","arxiv_id":"2503.21483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bolt-boost-large-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2503.21483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21483"}},"official":{"repos":["sming256/bolt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mobile-videogpt-fast-and-accurate-video","slug":"mobile-videogpt-fast-and-accurate-video","title":"Mobile-VideoGPT: Fast and Accurate Video Understanding Language Model","date":"2025-03-27","arxiv_id":"2503.21782","repositories_listed":1,"syntology":null},{"url":"/paper/acvubench-audio-centric-video-understanding","slug":"acvubench-audio-centric-video-understanding","title":"ACVUBench: Audio-Centric Video Understanding Benchmark","date":"2025-03-25","arxiv_id":"2503.19951","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrap-your-own-views-masked-ego-exo","slug":"bootstrap-your-own-views-masked-ego-exo","title":"Bootstrap Your Own Views: Masked Ego-Exo Modeling for Fine-grained View-invariant Video Representations","date":"2025-03-25","arxiv_id":"2503.19706","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-hallucination-of-large-multimodal","slug":"exploring-hallucination-of-large-multimodal","title":"Exploring Hallucination of Large Multimodal Models in Video Understanding: Benchmark, Analysis and Mitigation","date":"2025-03-25","arxiv_id":"2503.19622","repositories_listed":1,"syntology":null},{"url":"/paper/pave-patching-and-adapting-video-large","slug":"pave-patching-and-adapting-video-large","title":"PAVE: Patching and Adapting Video Large Language Models","date":"2025-03-25","arxiv_id":"2503.19794","repositories_listed":1,"syntology":null},{"url":"/paper/mammalps-a-multi-view-video-behavior","slug":"mammalps-a-multi-view-video-behavior","title":"MammAlps: A multi-view video behavior monitoring dataset of wild mammals in the Swiss Alps","date":"2025-03-23","arxiv_id":"2503.18223","repositories_listed":1,"syntology":null},{"url":"/paper/4d-bench-benchmarking-multi-modal-large","slug":"4d-bench-benchmarking-multi-modal-large","title":"4D-Bench: Benchmarking Multi-modal Large Language Models for 4D Object Understanding","date":"2025-03-22","arxiv_id":"2503.17827","repositories_listed":1,"syntology":null},{"url":"/paper/v2p-bench-evaluating-video-language","slug":"v2p-bench-evaluating-video-language","title":"V2P-Bench: Evaluating Video-Language Understanding with Visual Prompts for Better Human-Model Interaction","date":"2025-03-22","arxiv_id":"2503.17736","repositories_listed":1,"syntology":null},{"url":"/paper/agentic-keyframe-search-for-video-question","slug":"agentic-keyframe-search-for-video-question","title":"Agentic Keyframe Search for Video Question Answering","date":"2025-03-20","arxiv_id":"2503.16032","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-keyframe-search-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2503.16032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16032"}},"official":{"repos":["fansunqi/akeys"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-level-instruction-injection-for-video","slug":"hybrid-level-instruction-injection-for-video","title":"Hybrid-Level Instruction Injection for Video Token Compression in Multi-modal Large Language Models","date":"2025-03-20","arxiv_id":"2503.16036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-level-instruction-injection-for-video#ran","syntology_url":"https://syntology.ai/paper/2503.16036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16036"}},"official":{"repos":["lntzm/hicom"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-integrated-spatial-temporal-dynamic","slug":"stop-integrated-spatial-temporal-dynamic","title":"STOP: Integrated Spatial-Temporal Dynamic Prompting for Video Understanding","date":"2025-03-20","arxiv_id":"2503.15973","repositories_listed":1,"syntology":null},{"url":"/paper/xattention-block-sparse-attention-with","slug":"xattention-block-sparse-attention-with","title":"XAttention: Block Sparse Attention with Antidiagonal Scoring","date":"2025-03-20","arxiv_id":"2503.16428","repositories_listed":1,"syntology":null},{"url":"/paper/videomind-a-chain-of-lora-agent-for-long","slug":"videomind-a-chain-of-lora-agent-for-long","title":"VideoMind: A Chain-of-LoRA Agent for Long Video Reasoning","date":"2025-03-17","arxiv_id":"2503.13444","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomind-a-chain-of-lora-agent-for-long#ran","syntology_url":"https://syntology.ai/paper/2503.13444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13444"}},"official":{"repos":["yeliudev/VideoMind"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vispeak-visual-instruction-feedback-in","slug":"vispeak-visual-instruction-feedback-in","title":"ViSpeak: Visual Instruction Feedback in Streaming Videos","date":"2025-03-17","arxiv_id":"2503.12769","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vispeak-visual-instruction-feedback-in#ran","syntology_url":"https://syntology.ai/paper/2503.12769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12769"}},"official":null}},{"url":"/paper/adaretake-adaptive-redundancy-reduction-to","slug":"adaretake-adaptive-redundancy-reduction-to","title":"AdaReTaKe: Adaptive Redundancy Reduction to Perceive Longer for Video-language Understanding","date":"2025-03-16","arxiv_id":"2503.12559","repositories_listed":1,"syntology":{"n":23,"n_ran":17,"n_constructed":0,"n_ran_checked":15,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/adaretake-adaptive-redundancy-reduction-to#ran","syntology_url":"https://syntology.ai/paper/2503.12559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12559"}},"official":{"repos":["sczwangxiao/video-flexreduc"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/does-your-vision-language-model-get-lost-in","slug":"does-your-vision-language-model-get-lost-in","title":"Does Your Vision-Language Model Get Lost in the Long Video Sampling Dilemma?","date":"2025-03-16","arxiv_id":"2503.12496","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-your-vision-language-model-get-lost-in#ran","syntology_url":"https://syntology.ai/paper/2503.12496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12496"}},"official":{"repos":["dvlab-research/LSDBench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/keyframe-oriented-vision-token-pruning","slug":"keyframe-oriented-vision-token-pruning","title":"Keyframe-oriented Vision Token Pruning: Enhancing Efficiency of Large Vision Language Models on Long-Form Video Processing","date":"2025-03-13","arxiv_id":"2503.10742","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/keyframe-oriented-vision-token-pruning#ran","syntology_url":"https://syntology.ai/paper/2503.10742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10742"}},"official":{"repos":["1999Lyd/KVTP"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vlog-video-language-models-by-generative","slug":"vlog-video-language-models-by-generative","title":"VLog: Video-Language Models by Generative Retrieval of Narration Vocabulary","date":"2025-03-12","arxiv_id":"2503.09402","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vlog-video-language-models-by-generative#ran","syntology_url":"https://syntology.ai/paper/2503.09402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09402"}},"official":{"repos":["showlab/vlog"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/quota-query-oriented-token-assignment-via-cot","slug":"quota-query-oriented-token-assignment-via-cot","title":"QuoTA: Query-oriented Token Assignment via CoT Query Decouple for Long Video Comprehension","date":"2025-03-11","arxiv_id":"2503.08689","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quota-query-oriented-token-assignment-via-cot#ran","syntology_url":"https://syntology.ai/paper/2503.08689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08689"}},"official":{"repos":["mac-automl/quota"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/timeloc-a-unified-end-to-end-framework-for","slug":"timeloc-a-unified-end-to-end-framework-for","title":"TimeLoc: A Unified End-to-End Framework for Precise Timestamp Localization in Long Videos","date":"2025-03-09","arxiv_id":"2503.06526","repositories_listed":1,"syntology":null},{"url":"/paper/unified-reward-model-for-multimodal","slug":"unified-reward-model-for-multimodal","title":"Unified Reward Model for Multimodal Understanding and Generation","date":"2025-03-07","arxiv_id":"2503.05236","repositories_listed":1,"syntology":null},{"url":"/paper/egolife-towards-egocentric-life-assistant","slug":"egolife-towards-egocentric-life-assistant","title":"EgoLife: Towards Egocentric Life Assistant","date":"2025-03-05","arxiv_id":"2503.03803","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":9,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/egolife-towards-egocentric-life-assistant#ran","syntology_url":"https://syntology.ai/paper/2503.03803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03803"}},"official":{"repos":["evolvinglmms-lab/egolife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-fine-grained-hand-object-dynamics","slug":"modeling-fine-grained-hand-object-dynamics","title":"Modeling Fine-Grained Hand-Object Dynamics for Egocentric Video Representation Learning","date":"2025-03-02","arxiv_id":"2503.00986","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modeling-fine-grained-hand-object-dynamics#ran","syntology_url":"https://syntology.ai/paper/2503.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00986"}},"official":{"repos":["openrobotlab/egohod"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/opentad-a-unified-framework-and-comprehensive","slug":"opentad-a-unified-framework-and-comprehensive","title":"OpenTAD: A Unified Framework and Comprehensive Study of Temporal Action Detection","date":"2025-02-27","arxiv_id":"2502.20361","repositories_listed":1,"syntology":null},{"url":"/paper/task-graph-maximum-likelihood-estimation-for","slug":"task-graph-maximum-likelihood-estimation-for","title":"Task Graph Maximum Likelihood Estimation for Procedural Activity Understanding in Egocentric Videos","date":"2025-02-25","arxiv_id":"2502.17753","repositories_listed":1,"syntology":null},{"url":"/paper/video-salmonn-o1-reasoning-enhanced-audio","slug":"video-salmonn-o1-reasoning-enhanced-audio","title":"video-SALMONN-o1: Reasoning-enhanced Audio-visual Large Language Model","date":"2025-02-17","arxiv_id":"2502.11775","repositories_listed":1,"syntology":null},{"url":"/paper/vrope-rotary-position-embedding-for-video","slug":"vrope-rotary-position-embedding-for-video","title":"VRoPE: Rotary Position Embedding for Video Large Language Models","date":"2025-02-17","arxiv_id":"2502.11664","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vrope-rotary-position-embedding-for-video#ran","syntology_url":"https://syntology.ai/paper/2502.11664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11664"}},"official":{"repos":["johncaged/vrope"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/svbench-a-benchmark-with-temporal-multi-turn","slug":"svbench-a-benchmark-with-temporal-multi-turn","title":"SVBench: A Benchmark with Temporal Multi-Turn Dialogues for Streaming Video Understanding","date":"2025-02-15","arxiv_id":"2502.10810","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/svbench-a-benchmark-with-temporal-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2502.10810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10810"}},"official":{"repos":["yzy-bupt/SVBench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/long-vita-scaling-large-multi-modal-models-to","slug":"long-vita-scaling-large-multi-modal-models-to","title":"Long-VITA: Scaling Large Multi-modal Models to 1 Million Tokens with Leading Short-Context Accuray","date":"2025-02-07","arxiv_id":"2502.05177","repositories_listed":1,"syntology":null},{"url":"/paper/videorope-what-makes-for-good-video-rotary","slug":"videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","arxiv_id":"2502.05173","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorope-what-makes-for-good-video-rotary#ran","syntology_url":"https://syntology.ai/paper/2502.05173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05173"}},"official":{"repos":["wiselnn570/videorope"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hier-egopack-hierarchical-egocentric-video","slug":"hier-egopack-hierarchical-egocentric-video","title":"Hier-EgoPack: Hierarchical Egocentric Video Understanding with Diverse Task Perspectives","date":"2025-02-04","arxiv_id":"2502.02487","repositories_listed":1,"syntology":null},{"url":"/paper/tumtraffic-videoqa-a-benchmark-for-unified","slug":"tumtraffic-videoqa-a-benchmark-for-unified","title":"TUMTraffic-VideoQA: A Benchmark for Unified Spatio-Temporal Video Understanding in Traffic Scenes","date":"2025-02-04","arxiv_id":"2502.02449","repositories_listed":1,"syntology":null},{"url":"/paper/videorag-retrieval-augmented-generation-with","slug":"videorag-retrieval-augmented-generation-with","title":"VideoRAG: Retrieval-Augmented Generation with Extreme Long-Context Videos","date":"2025-02-03","arxiv_id":"2502.01549","repositories_listed":1,"syntology":null},{"url":"/paper/ain-the-arabic-inclusive-large-multimodal","slug":"ain-the-arabic-inclusive-large-multimodal","title":"AIN: The Arabic INclusive Large Multimodal Model","date":"2025-01-31","arxiv_id":"2502.00094","repositories_listed":1,"syntology":null},{"url":"/paper/infty-video-a-training-free-approach-to-long","slug":"infty-video-a-training-free-approach-to-long","title":"$\\infty$-Video: A Training-Free Approach to Long Video Understanding via Continuous-Time Memory Consolidation","date":"2025-01-31","arxiv_id":"2501.19098","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infty-video-a-training-free-approach-to-long#ran","syntology_url":"https://syntology.ai/paper/2501.19098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19098"}},"official":{"repos":["deep-spin/infinite-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tinyllava-video-a-simple-framework-of-small","slug":"tinyllava-video-a-simple-framework-of-small","title":"TinyLLaVA-Video: A Simple Framework of Small-scale Large Multimodal Models for Video Understanding","date":"2025-01-26","arxiv_id":"2501.15513","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-video-understanding-and-multi-round","slug":"streaming-video-understanding-and-multi-round","title":"Streaming Video Understanding and Multi-round Interaction with Memory-enhanced Knowledge","date":"2025-01-23","arxiv_id":"2501.13468","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/streaming-video-understanding-and-multi-round#ran","syntology_url":"https://syntology.ai/paper/2501.13468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13468"}},"official":{"repos":["hmxiong/streamchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-3-frontier-multimodal-foundation","slug":"videollama-3-frontier-multimodal-foundation","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","date":"2025-01-22","arxiv_id":"2501.13106","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/videollama-3-frontier-multimodal-foundation#ran","syntology_url":"https://syntology.ai/paper/2501.13106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13106"}},"official":{"repos":["damo-nlp-sg/videollama3"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/internlm-xcomposer2-5-reward-a-simple-yet","slug":"internlm-xcomposer2-5-reward-a-simple-yet","title":"InternLM-XComposer2.5-Reward: A Simple Yet Effective Multi-Modal Reward Model","date":"2025-01-21","arxiv_id":"2501.12368","repositories_listed":1,"syntology":null},{"url":"/paper/internvideo2-5-empowering-video-mllms-with","slug":"internvideo2-5-empowering-video-mllms-with","title":"InternVideo2.5: Empowering Video MLLMs with Long and Rich Context Modeling","date":"2025-01-21","arxiv_id":"2501.12386","repositories_listed":1,"syntology":null},{"url":"/paper/mmvu-measuring-expert-level-multi-discipline","slug":"mmvu-measuring-expert-level-multi-discipline","title":"MMVU: Measuring Expert-Level Multi-Discipline Video Understanding","date":"2025-01-21","arxiv_id":"2501.12380","repositories_listed":1,"syntology":null},{"url":"/paper/avs-mamba-exploring-temporal-and-multi-modal","slug":"avs-mamba-exploring-temporal-and-multi-modal","title":"AVS-Mamba: Exploring Temporal and Multi-modal Mamba for Audio-Visual Segmentation","date":"2025-01-14","arxiv_id":"2501.07810","repositories_listed":1,"syntology":null},{"url":"/paper/facial-dynamics-in-video-instruction-tuning","slug":"facial-dynamics-in-video-instruction-tuning","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","date":"2025-01-14","arxiv_id":"2501.07978","repositories_listed":1,"syntology":null},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mecd-unlocking-event-level-causal-graph","slug":"mecd-unlocking-event-level-causal-graph","title":"MECD+: Unlocking Event-Level Causal Graph Discovery for Video Reasoning","date":"2025-01-13","arxiv_id":"2501.07227","repositories_listed":1,"syntology":null}],"record_sha256":"e41bcaeba7860bb6d9b9bd159be2130435241e6c70524e8b55bf683c54ff8c4e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}