{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/3","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":12,"rows_per_page":100,"rows":[201,300],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding/papers/2","next":"/task/video-understanding/papers/4","papers":[{"url":"/paper/vidchain-chain-of-tasks-with-metric-based","slug":"vidchain-chain-of-tasks-with-metric-based","title":"VidChain: Chain-of-Tasks with Metric-based Direct Preference Optimization for Dense Video Captioning","date":"2025-01-12","arxiv_id":"2501.06761","repositories_listed":1,"syntology":null},{"url":"/paper/from-my-view-to-yours-ego-augmented-learning","slug":"from-my-view-to-yours-ego-augmented-learning","title":"From My View to Yours: Ego-Augmented Learning in Large Vision Language Models for Understanding Exocentric Daily Living Activities","date":"2025-01-10","arxiv_id":"2501.05711","repositories_listed":1,"syntology":null},{"url":"/paper/valley2-exploring-multimodal-models-with","slug":"valley2-exploring-multimodal-models-with","title":"Valley2: Exploring Multimodal Models with Scalable Vision-Language Design","date":"2025-01-10","arxiv_id":"2501.05901","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/valley2-exploring-multimodal-models-with#ran","syntology_url":"https://syntology.ai/paper/2501.05901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05901"}},"official":{"repos":["bytedance/valley"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ovo-bench-how-far-is-your-video-llms-from","slug":"ovo-bench-how-far-is-your-video-llms-from","title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","date":"2025-01-09","arxiv_id":"2501.05510","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ovo-bench-how-far-is-your-video-llms-from#ran","syntology_url":"https://syntology.ai/paper/2501.05510","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05510"}},"official":{"repos":["joeleelyf/ovo-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hlv-1k-a-large-scale-hour-long-video","slug":"hlv-1k-a-large-scale-hour-long-video","title":"HLV-1K: A Large-scale Hour-Long Video Benchmark for Time-Specific Long Video Understanding","date":"2025-01-03","arxiv_id":"2501.01645","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-specialized-visual-encoders-for","slug":"unifying-specialized-visual-encoders-for","title":"Unifying Specialized Visual Encoders for Video Language Models","date":"2025-01-02","arxiv_id":"2501.01426","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-specialized-visual-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2501.01426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01426"}},"official":{"repos":["princetonvisualai/merv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-keyframe-sampling-for-long-video","slug":"adaptive-keyframe-sampling-for-long-video","title":"Adaptive Keyframe Sampling for Long Video Understanding","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/language-guided-audio-visual-learning-for","slug":"language-guided-audio-visual-learning-for","title":"Language-Guided Audio-Visual Learning for Long-Term Sports Assessment","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mamba4d-efficient-4d-point-cloud-video","slug":"mamba4d-efficient-4d-point-cloud-video","title":"Mamba4D: Efficient 4D Point Cloud Video Understanding with Disentangled Spatial-Temporal State Space Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/online-video-understanding-a-comprehensive","slug":"online-video-understanding-a-comprehensive","title":"Online Video Understanding: OVBench and VideoChat-Online","date":"2024-12-31","arxiv_id":"2501.00584","repositories_listed":1,"syntology":null},{"url":"/paper/videorefer-suite-advancing-spatial-temporal","slug":"videorefer-suite-advancing-spatial-temporal","title":"VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM","date":"2024-12-31","arxiv_id":"2501.00599","repositories_listed":1,"syntology":null},{"url":"/paper/detection-fusion-for-knowledge-graph","slug":"detection-fusion-for-knowledge-graph","title":"Detection-Fusion for Knowledge Graph Extraction from Videos","date":"2024-12-30","arxiv_id":"2501.00136","repositories_listed":1,"syntology":null},{"url":"/paper/framefusion-combining-similarity-and","slug":"framefusion-combining-similarity-and","title":"FrameFusion: Combining Similarity and Importance for Video Token Reduction on Large Visual Language Models","date":"2024-12-30","arxiv_id":"2501.01986","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/framefusion-combining-similarity-and#ran","syntology_url":"https://syntology.ai/paper/2501.01986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01986"}},"official":{"repos":["thu-nics/framefusion"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/retake-reducing-temporal-and-knowledge","slug":"retake-reducing-temporal-and-knowledge","title":"ReTaKe: Reducing Temporal and Knowledge Redundancy for Long Video Understanding","date":"2024-12-29","arxiv_id":"2412.20504","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/retake-reducing-temporal-and-knowledge#ran","syntology_url":"https://syntology.ai/paper/2412.20504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.20504"}},"official":{"repos":["sczwangxiao/video-retake"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/humanvbench-exploring-human-centric-video","slug":"humanvbench-exploring-human-centric-video","title":"HumanVBench: Exploring Human-Centric Video Understanding Capabilities of MLLMs with Synthetic Benchmark Data","date":"2024-12-23","arxiv_id":"2412.17574","repositories_listed":1,"syntology":null},{"url":"/paper/friendsqa-a-new-large-scale-deep-video","slug":"friendsqa-a-new-large-scale-deep-video","title":"FriendsQA: A New Large-Scale Deep Video Understanding Dataset with Fine-grained Topic Categorization for Story Videos","date":"2024-12-22","arxiv_id":"2412.17022","repositories_listed":1,"syntology":null},{"url":"/paper/prunevid-visual-token-pruning-for-efficient","slug":"prunevid-visual-token-pruning-for-efficient","title":"PruneVid: Visual Token Pruning for Efficient Video Large Language Models","date":"2024-12-20","arxiv_id":"2412.16117","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/prunevid-visual-token-pruning-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2412.16117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16117"}},"official":{"repos":["visual-ai/prunevid"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/do-language-models-understand-time","slug":"do-language-models-understand-time","title":"Do Language Models Understand Time?","date":"2024-12-18","arxiv_id":"2412.13845","repositories_listed":1,"syntology":null},{"url":"/paper/flashvtg-feature-layering-and-adaptive-score","slug":"flashvtg-feature-layering-and-adaptive-score","title":"FlashVTG: Feature Layering and Adaptive Score Handling Network for Video Temporal Grounding","date":"2024-12-18","arxiv_id":"2412.13441","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flashvtg-feature-layering-and-adaptive-score#ran","syntology_url":"https://syntology.ai/paper/2412.13441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13441"}},"official":{"repos":["zhuo-cao/flashvtg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/instructseg-unifying-instructed-visual","slug":"instructseg-unifying-instructed-visual","title":"InstructSeg: Unifying Instructed Visual Segmentation with Multi-modal Large Language Models","date":"2024-12-18","arxiv_id":"2412.14006","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/instructseg-unifying-instructed-visual#ran","syntology_url":"https://syntology.ai/paper/2412.14006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14006"}},"official":{"repos":["congvvc/instructseg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-adafocus-spatial-temporal-dynamic","slug":"uni-adafocus-spatial-temporal-dynamic","title":"Uni-AdaFocus: Spatial-temporal Dynamic Computation for Video Recognition","date":"2024-12-15","arxiv_id":"2412.11228","repositories_listed":1,"syntology":null},{"url":"/paper/b-vllm-a-vision-large-language-model-with","slug":"b-vllm-a-vision-large-language-model-with","title":"B-VLLM: A Vision Large Language Model with Balanced Spatio-Temporal Tokens","date":"2024-12-13","arxiv_id":"2412.09919","repositories_listed":1,"syntology":null},{"url":"/paper/neptune-the-long-orbit-to-benchmarking-long","slug":"neptune-the-long-orbit-to-benchmarking-long","title":"Neptune: The Long Orbit to Benchmarking Long Video Understanding","date":"2024-12-12","arxiv_id":"2412.09582","repositories_listed":1,"syntology":null},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/linvt-empower-your-image-level-large-language","slug":"linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","arxiv_id":"2412.05185","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":6,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/linvt-empower-your-image-level-large-language#ran","syntology_url":"https://syntology.ai/paper/2412.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05185"}},"official":{"repos":["gls0425/linvt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/visionzip-longer-is-better-but-not-necessary","slug":"visionzip-longer-is-better-but-not-necessary","title":"VisionZip: Longer is Better but Not Necessary in Vision Language Models","date":"2024-12-05","arxiv_id":"2412.04467","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visionzip-longer-is-better-but-not-necessary#ran","syntology_url":"https://syntology.ai/paper/2412.04467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04467"}},"official":{"repos":["dvlab-research/visionzip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/aim-adaptive-inference-of-multi-modal-llms","slug":"aim-adaptive-inference-of-multi-modal-llms","title":"AIM: Adaptive Inference of Multi-Modal LLMs via Token Merging and Pruning","date":"2024-12-04","arxiv_id":"2412.03248","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/aim-adaptive-inference-of-multi-modal-llms#ran","syntology_url":"https://syntology.ai/paper/2412.03248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03248"}},"official":{"repos":["lavi-lab/aim"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/inst-it-boosting-multimodal-instance","slug":"inst-it-boosting-multimodal-instance","title":"Inst-IT: Boosting Multimodal Instance Understanding via Explicit Visual Prompt Instruction Tuning","date":"2024-12-04","arxiv_id":"2412.03565","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inst-it-boosting-multimodal-instance#ran","syntology_url":"https://syntology.ai/paper/2412.03565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03565"}},"official":{"repos":["inst-it/inst-it"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-detection-of-queried-event-start","slug":"streaming-detection-of-queried-event-start","title":"Streaming Detection of Queried Event Start","date":"2024-12-04","arxiv_id":"2412.03567","repositories_listed":1,"syntology":null},{"url":"/paper/videoicl-confidence-based-iterative-in","slug":"videoicl-confidence-based-iterative-in","title":"VideoICL: Confidence-based Iterative In-context Learning for Out-of-Distribution Video Understanding","date":"2024-12-03","arxiv_id":"2412.02186","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videoicl-confidence-based-iterative-in#ran","syntology_url":"https://syntology.ai/paper/2412.02186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02186"}},"official":{"repos":["kangsankim07/videoicl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/physgame-uncovering-physical-commonsense-1","slug":"physgame-uncovering-physical-commonsense-1","title":"PhysGame: Uncovering Physical Commonsense Violations in Gameplay Videos","date":"2024-12-02","arxiv_id":"2412.01800","repositories_listed":1,"syntology":null},{"url":"/paper/towards-universal-soccer-video-understanding","slug":"towards-universal-soccer-video-understanding","title":"Towards Universal Soccer Video Understanding","date":"2024-12-02","arxiv_id":"2412.01820","repositories_listed":1,"syntology":null},{"url":"/paper/longvale-vision-audio-language-event","slug":"longvale-vision-audio-language-event","title":"LongVALE: Vision-Audio-Language-Event Benchmark Towards Time-Aware Omni-Modal Perception of Long Videos","date":"2024-11-29","arxiv_id":"2411.19772","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/longvale-vision-audio-language-event#ran","syntology_url":"https://syntology.ai/paper/2411.19772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19772"}},"official":{"repos":["ttgeng233/LongVALE"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/t2vid-translating-long-text-into-multi-image","slug":"t2vid-translating-long-text-into-multi-image","title":"T2Vid: Translating Long Text into Multi-Image is the Catalyst for Video-LLMs","date":"2024-11-29","arxiv_id":"2411.19951","repositories_listed":1,"syntology":null},{"url":"/paper/timemarker-a-versatile-video-llm-for-long-and","slug":"timemarker-a-versatile-video-llm-for-long-and","title":"TimeMarker: A Versatile Video-LLM for Long and Short Video Understanding with Superior Temporal Localization Ability","date":"2024-11-27","arxiv_id":"2411.18211","repositories_listed":1,"syntology":null},{"url":"/paper/occludenet-a-causal-journey-into-mixed-view","slug":"occludenet-a-causal-journey-into-mixed-view","title":"OccludeNet: A Causal Journey into Mixed-View Actor-Centric Video Action Recognition under Occlusions","date":"2024-11-24","arxiv_id":"2411.15729","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-vlms-to-localize-specific-objects","slug":"teaching-vlms-to-localize-specific-objects","title":"Teaching VLMs to Localize Specific Objects from In-context Examples","date":"2024-11-20","arxiv_id":"2411.13317","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-vlms-to-localize-specific-objects#ran","syntology_url":"https://syntology.ai/paper/2411.13317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13317"}},"official":{"repos":["sivandoveh/iploc"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-rag-visually-aligned-retrieval","slug":"video-rag-visually-aligned-retrieval","title":"Video-RAG: Visually-aligned Retrieval-Augmented Long Video Comprehension","date":"2024-11-20","arxiv_id":"2411.13093","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-rag-visually-aligned-retrieval#ran","syntology_url":"https://syntology.ai/paper/2411.13093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13093"}},"official":{"repos":["leon1207/video-rag-master"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/ts-llava-constructing-visual-tokens-through","slug":"ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","arxiv_id":"2411.11066","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ts-llava-constructing-visual-tokens-through#ran","syntology_url":"https://syntology.ai/paper/2411.11066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11066"}},"official":{"repos":["tingyu215/ts-llava"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/streamingbench-assessing-the-gap-for-mllms-to","slug":"streamingbench-assessing-the-gap-for-mllms-to","title":"StreamingBench: Assessing the Gap for MLLMs to Achieve Streaming Video Understanding","date":"2024-11-06","arxiv_id":"2411.03628","repositories_listed":1,"syntology":null},{"url":"/paper/ppllava-varied-video-sequence-understanding","slug":"ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ppllava-varied-video-sequence-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.02327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02327"}},"official":{"repos":["farewellthree/ppllava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/language-assisted-skeleton-action","slug":"language-assisted-skeleton-action","title":"Language-Assisted Skeleton Action Understanding for Skeleton-Based Temporal Action Segmentation","date":"2024-10-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/situational-scene-graph-for-structured-human","slug":"situational-scene-graph-for-structured-human","title":"Situational Scene Graph for Structured Human-centric Situation Understanding","date":"2024-10-30","arxiv_id":"2410.22829","repositories_listed":1,"syntology":null},{"url":"/paper/tomato-assessing-visual-temporal-reasoning","slug":"tomato-assessing-visual-temporal-reasoning","title":"TOMATO: Assessing Visual Temporal Reasoning Capabilities in Multimodal Foundation Models","date":"2024-10-30","arxiv_id":"2410.23266","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tomato-assessing-visual-temporal-reasoning#ran","syntology_url":"https://syntology.ai/paper/2410.23266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23266"}},"official":{"repos":["yale-nlp/TOMATO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timesuite-improving-mllms-for-long-video","slug":"timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","arxiv_id":"2410.19702","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timesuite-improving-mllms-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2410.19702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19702"}},"official":null}},{"url":"/paper/camel-bench-a-comprehensive-arabic-lmm","slug":"camel-bench-a-comprehensive-arabic-lmm","title":"CAMEL-Bench: A Comprehensive Arabic LMM Benchmark","date":"2024-10-24","arxiv_id":"2410.18976","repositories_listed":1,"syntology":null},{"url":"/paper/videowebarena-evaluating-long-context","slug":"videowebarena-evaluating-long-context","title":"VideoWebArena: Evaluating Long Context Multimodal Agents with Video Understanding Web Tasks","date":"2024-10-24","arxiv_id":"2410.19100","repositories_listed":1,"syntology":null},{"url":"/paper/longvu-spatiotemporal-adaptive-compression","slug":"longvu-spatiotemporal-adaptive-compression","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","date":"2024-10-22","arxiv_id":"2410.17434","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvu-spatiotemporal-adaptive-compression#ran","syntology_url":"https://syntology.ai/paper/2410.17434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17434"}},"official":{"repos":["Vision-CAIR/LongVU"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videgothink-assessing-egocentric-video","slug":"videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","arxiv_id":"2410.11623","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":1,"n_no_contract":9,"n_pointer_only":5,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videgothink-assessing-egocentric-video#ran","syntology_url":"https://syntology.ai/paper/2410.11623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11623"}},"official":null}},{"url":"/paper/free-video-llm-prompt-guided-visual","slug":"free-video-llm-prompt-guided-visual","title":"Free Video-LLM: Prompt-guided Visual Perception for Efficient Training-free Video LLMs","date":"2024-10-14","arxiv_id":"2410.10441","repositories_listed":1,"syntology":null},{"url":"/paper/temporalbench-benchmarking-fine-grained","slug":"temporalbench-benchmarking-fine-grained","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","date":"2024-10-14","arxiv_id":"2410.10818","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporalbench-benchmarking-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.10818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10818"}},"official":{"repos":["mu-cai/TemporalBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/verified-a-video-corpus-moment-retrieval","slug":"verified-a-video-corpus-moment-retrieval","title":"VERIFIED: A Video Corpus Moment Retrieval Benchmark for Fine-Grained Video Understanding","date":"2024-10-11","arxiv_id":"2410.08593","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-temporal-modeling-of-video-llms-via","slug":"enhancing-temporal-modeling-of-video-llms-via","title":"Enhancing Temporal Modeling of Video LLMs via Time Gating","date":"2024-10-08","arxiv_id":"2410.05714","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/enhancing-temporal-modeling-of-video-llms-via#ran","syntology_url":"https://syntology.ai/paper/2410.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05714"}},"official":{"repos":["lavi-lab/tg-vid"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/trace-temporal-grounding-video-llm-via-causal","slug":"trace-temporal-grounding-video-llm-via-causal","title":"TRACE: Temporal Grounding Video LLM via Causal Event Modeling","date":"2024-10-08","arxiv_id":"2410.05643","repositories_listed":1,"syntology":{"n":16,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/trace-temporal-grounding-video-llm-via-causal#ran","syntology_url":"https://syntology.ai/paper/2410.05643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05643"}},"official":{"repos":["gyxxyg/trace"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/sparsevlm-visual-token-sparsification-for","slug":"sparsevlm-visual-token-sparsification-for","title":"SparseVLM: Visual Token Sparsification for Efficient Vision-Language Model Inference","date":"2024-10-06","arxiv_id":"2410.04417","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sparsevlm-visual-token-sparsification-for#ran","syntology_url":"https://syntology.ai/paper/2410.04417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04417"}},"official":{"repos":["gumpest/sparsevlms"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-videollm-sharpening-fine-grained","slug":"grounded-videollm-sharpening-fine-grained","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","date":"2024-10-04","arxiv_id":"2410.03290","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-videollm-sharpening-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.03290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03290"}},"official":{"repos":["whb139426/grounded-video-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ual-bench-the-first-comprehensive-unusual","slug":"ual-bench-the-first-comprehensive-unusual","title":"UAL-Bench: The First Comprehensive Unusual Activity Localization Benchmark","date":"2024-10-02","arxiv_id":"2410.01180","repositories_listed":1,"syntology":null},{"url":"/paper/scvlm-a-vision-language-model-for-driving","slug":"scvlm-a-vision-language-model-for-driving","title":"ScVLM: Enhancing Vision-Language Model for Safety-Critical Event Understanding","date":"2024-10-01","arxiv_id":"2410.00982","repositories_listed":1,"syntology":null},{"url":"/paper/videoinsta-zero-shot-long-video-understanding","slug":"videoinsta-zero-shot-long-video-understanding","title":"VideoINSTA: Zero-shot Long Video Understanding via Informative Spatial-Temporal Reasoning with LLMs","date":"2024-09-30","arxiv_id":"2409.20365","repositories_listed":1,"syntology":null},{"url":"/paper/from-seconds-to-hours-reviewing-multimodal","slug":"from-seconds-to-hours-reviewing-multimodal","title":"From Seconds to Hours: Reviewing MultiModal Large Language Models on Comprehensive Long Video Understanding","date":"2024-09-27","arxiv_id":"2409.18938","repositories_listed":1,"syntology":null},{"url":"/paper/e-t-bench-towards-open-ended-event-level","slug":"e-t-bench-towards-open-ended-event-level","title":"E.T. Bench: Towards Open-Ended Event-Level Video-Language Understanding","date":"2024-09-26","arxiv_id":"2409.18111","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/e-t-bench-towards-open-ended-event-level#ran","syntology_url":"https://syntology.ai/paper/2409.18111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18111"}},"official":{"repos":["PolyU-ChenLab/ETBench"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/video-xl-extra-long-vision-language-model-for","slug":"video-xl-extra-long-vision-language-model-for","title":"Video-XL: Extra-Long Vision Language Model for Hour-Scale Video Understanding","date":"2024-09-22","arxiv_id":"2409.14485","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-xl-extra-long-vision-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2409.14485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14485"}},"official":{"repos":["vectorspacelab/video-xl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/soccernet-2024-challenges-results","slug":"soccernet-2024-challenges-results","title":"SoccerNet 2024 Challenges Results","date":"2024-09-16","arxiv_id":"2409.10587","repositories_listed":1,"syntology":null},{"url":"/paper/longllava-scaling-multi-modal-llms-to-1000","slug":"longllava-scaling-multi-modal-llms-to-1000","title":"LongLLaVA: Scaling Multi-modal LLMs to 1000 Images Efficiently via a Hybrid Architecture","date":"2024-09-04","arxiv_id":"2409.02889","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longllava-scaling-multi-modal-llms-to-1000#ran","syntology_url":"https://syntology.ai/paper/2409.02889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02889"}},"official":{"repos":["freedomintelligence/longllava"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-ccam-enhancing-video-language","slug":"video-ccam-enhancing-video-language","title":"Video-CCAM: Enhancing Video-Language Understanding with Causal Cross-Attention Masks for Short and Long Videos","date":"2024-08-26","arxiv_id":"2408.14023","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-ccam-enhancing-video-language#ran","syntology_url":"https://syntology.ai/paper/2408.14023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14023"}},"official":{"repos":["qq-mm/video-ccam"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longvila-scaling-long-context-visual-language","slug":"longvila-scaling-long-context-visual-language","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","date":"2024-08-19","arxiv_id":"2408.10188","repositories_listed":1,"syntology":null},{"url":"/paper/hat-history-augmented-anchor-transformer-for","slug":"hat-history-augmented-anchor-transformer-for","title":"HAT: History-Augmented Anchor Transformer for Online Temporal Action Localization","date":"2024-08-12","arxiv_id":"2408.06437","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":3,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 3 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hat-history-augmented-anchor-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2408.06437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06437"}},"official":{"repos":["sakibreza/eccv24-hat"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":3,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videoqa-in-the-era-of-llms-an-empirical-study","slug":"videoqa-in-the-era-of-llms-an-empirical-study","title":"VideoQA in the Era of LLMs: An Empirical Study","date":"2024-08-08","arxiv_id":"2408.04223","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02272","slug":"2408-02272","title":"COM Kitchens: An Unedited Overhead-view Video Dataset as a Vision-Language Benchmark","date":"2024-08-05","arxiv_id":"2408.02272","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-temporal-causality-for-advanced","slug":"harnessing-temporal-causality-for-advanced","title":"Harnessing Temporal Causality for Advanced Temporal Action Detection","date":"2024-07-25","arxiv_id":"2407.17792","repositories_listed":1,"syntology":null},{"url":"/paper/egocvr-an-egocentric-benchmark-for-fine","slug":"egocvr-an-egocentric-benchmark-for-fine","title":"EgoCVR: An Egocentric Benchmark for Fine-Grained Composed Video Retrieval","date":"2024-07-23","arxiv_id":"2407.16658","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/egocvr-an-egocentric-benchmark-for-fine#ran","syntology_url":"https://syntology.ai/paper/2407.16658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16658"}},"official":{"repos":["explainableml/egocvr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/longvideobench-a-benchmark-for-long-context","slug":"longvideobench-a-benchmark-for-long-context","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","date":"2024-07-22","arxiv_id":"2407.15754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvideobench-a-benchmark-for-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.15754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15754"}},"official":{"repos":["longvideobench/longvideobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slowfast-llava-a-strong-training-free","slug":"slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","arxiv_id":"2407.15841","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slowfast-llava-a-strong-training-free#ran","syntology_url":"https://syntology.ai/paper/2407.15841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15841"}},"official":{"repos":["apple/ml-slowfast-llava"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/goldfish-vision-language-understanding-of","slug":"goldfish-vision-language-understanding-of","title":"Goldfish: Vision-Language Understanding of Arbitrarily Long Videos","date":"2024-07-17","arxiv_id":"2407.12679","repositories_listed":1,"syntology":null},{"url":"/paper/hypergraph-multi-modal-large-language-model","slug":"hypergraph-multi-modal-large-language-model","title":"Hypergraph Multi-modal Large Language Model: Exploiting EEG and Eye-tracking Modalities to Evaluate Heterogeneous Responses for Video Understanding","date":"2024-07-11","arxiv_id":"2407.08150","repositories_listed":1,"syntology":null},{"url":"/paper/videomamba-spatio-temporal-selective-state","slug":"videomamba-spatio-temporal-selective-state","title":"VideoMamba: Spatio-Temporal Selective State Space Model","date":"2024-07-11","arxiv_id":"2407.08476","repositories_listed":1,"syntology":null},{"url":"/paper/videoeval-comprehensive-benchmark-suite-for","slug":"videoeval-comprehensive-benchmark-suite-for","title":"VideoEval: Comprehensive Benchmark Suite for Low-Cost Evaluation of Video Foundation Model","date":"2024-07-09","arxiv_id":"2407.06491","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/videoeval-comprehensive-benchmark-suite-for#ran","syntology_url":"https://syntology.ai/paper/2407.06491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06491"}},"official":{"repos":["leexinhao/VideoEval"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/video-star-self-training-enables-video","slug":"video-star-self-training-enables-video","title":"Video-STaR: Self-Training Enables Video Instruction Tuning with Any Supervision","date":"2024-07-08","arxiv_id":"2407.06189","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-star-self-training-enables-video#ran","syntology_url":"https://syntology.ai/paper/2407.06189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06189"}},"official":{"repos":["orrzohar/Video-STaR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmad-multi-label-micro-action-detection-in","slug":"mmad-multi-label-micro-action-detection-in","title":"MMAD: Multi-label Micro-Action Detection in Videos","date":"2024-07-07","arxiv_id":"2407.05311","repositories_listed":1,"syntology":null},{"url":"/paper/internlm-xcomposer-2-5-a-versatile-large","slug":"internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","arxiv_id":"2407.03320","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internlm-xcomposer-2-5-a-versatile-large#ran","syntology_url":"https://syntology.ai/paper/2407.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03320"}},"official":{"repos":["internlm/internlm-xcomposer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/tarsier-recipes-for-training-and-evaluating-1","slug":"tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","arxiv_id":"2407.00634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier-recipes-for-training-and-evaluating-1#ran","syntology_url":"https://syntology.ai/paper/2407.00634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00634"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infinibench-a-comprehensive-benchmark-for","slug":"infinibench-a-comprehensive-benchmark-for","title":"InfiniBench: A Comprehensive Benchmark for Large Multimodal Models in Very Long Video Understanding","date":"2024-06-28","arxiv_id":"2406.19875","repositories_listed":1,"syntology":null},{"url":"/paper/fibottention-inceptive-visual-representation","slug":"fibottention-inceptive-visual-representation","title":"Fibottention: Inceptive Visual Representation Learning with Diverse Attention Across Heads","date":"2024-06-27","arxiv_id":"2406.19391","repositories_listed":1,"syntology":null},{"url":"/paper/omg-llava-bridging-image-level-object-level","slug":"omg-llava-bridging-image-level-object-level","title":"OMG-LLaVA: Bridging Image-level, Object-level, Pixel-level Reasoning and Understanding","date":"2024-06-27","arxiv_id":"2406.19389","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omg-llava-bridging-image-level-object-level#ran","syntology_url":"https://syntology.ai/paper/2406.19389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19389"}},"official":null}},{"url":"/paper/videomambapro-a-leap-forward-for-mamba-in","slug":"videomambapro-a-leap-forward-for-mamba-in","title":"Snakes and Ladders: Two Steps Up for VideoMamba","date":"2024-06-27","arxiv_id":"2406.19006","repositories_listed":1,"syntology":null},{"url":"/paper/mmbench-video-a-long-form-multi-shot","slug":"mmbench-video-a-long-form-multi-shot","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","date":"2024-06-20","arxiv_id":"2406.14515","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmbench-video-a-long-form-multi-shot#ran","syntology_url":"https://syntology.ai/paper/2406.14515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14515"}},"official":{"repos":["open-compass/vlmevalkit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-event-oriented-long-video","slug":"towards-event-oriented-long-video","title":"Towards Event-oriented Long Video Understanding","date":"2024-06-20","arxiv_id":"2406.14129","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-event-oriented-long-video#ran","syntology_url":"https://syntology.ai/paper/2406.14129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14129"}},"official":{"repos":["rucaibox/event-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slot-state-space-models","slug":"slot-state-space-models","title":"Slot State Space Models","date":"2024-06-18","arxiv_id":"2406.12272","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slot-state-space-models#ran","syntology_url":"https://syntology.ai/paper/2406.12272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12272"}},"official":{"repos":["jindongjiang/slotssms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videovista-a-versatile-benchmark-for-video","slug":"videovista-a-versatile-benchmark-for-video","title":"VideoVista: A Versatile Benchmark for Video Understanding and Reasoning","date":"2024-06-17","arxiv_id":"2406.11303","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-raw-videos-understanding-edited-videos","slug":"beyond-raw-videos-understanding-edited-videos","title":"Beyond Raw Videos: Understanding Edited Videos with Large Multimodal Model","date":"2024-06-15","arxiv_id":"2406.10484","repositories_listed":1,"syntology":null},{"url":"/paper/needle-in-a-video-haystack-a-scalable","slug":"needle-in-a-video-haystack-a-scalable","title":"Needle In A Video Haystack: A Scalable Synthetic Evaluator for Video MLLMs","date":"2024-06-13","arxiv_id":"2406.09367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/needle-in-a-video-haystack-a-scalable#ran","syntology_url":"https://syntology.ai/paper/2406.09367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09367"}},"official":{"repos":["joez17/videoniah"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videogpt-integrating-image-and-video-encoders#ran","syntology_url":"https://syntology.ai/paper/2406.09418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09418"}},"official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flash-vstream-memory-based-real-time","slug":"flash-vstream-memory-based-real-time","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","date":"2024-06-12","arxiv_id":"2406.08085","repositories_listed":1,"syntology":null},{"url":"/paper/lvbench-an-extreme-long-video-understanding","slug":"lvbench-an-extreme-long-video-understanding","title":"LVBench: An Extreme Long Video Understanding Benchmark","date":"2024-06-12","arxiv_id":"2406.08035","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvbench-an-extreme-long-video-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.08035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08035"}},"official":{"repos":["THUDM/LVBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmworld-towards-multi-discipline-multi","slug":"mmworld-towards-multi-discipline-multi","title":"MMWorld: Towards Multi-discipline Multi-faceted World Model Evaluation in Videos","date":"2024-06-12","arxiv_id":"2406.08407","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmworld-towards-multi-discipline-multi#ran","syntology_url":"https://syntology.ai/paper/2406.08407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08407"}},"official":{"repos":["eric-ai-lab/mmworld"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vript-a-video-is-worth-thousands-of-words","slug":"vript-a-video-is-worth-thousands-of-words","title":"Vript: A Video Is Worth Thousands of Words","date":"2024-06-10","arxiv_id":"2406.06040","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vript-a-video-is-worth-thousands-of-words#ran","syntology_url":"https://syntology.ai/paper/2406.06040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06040"}},"official":{"repos":["mutonix/vript"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sharegpt4video-improving-video-understanding","slug":"sharegpt4video-improving-video-understanding","title":"ShareGPT4Video: Improving Video Understanding and Generation with Better Captions","date":"2024-06-06","arxiv_id":"2406.04325","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-task-graph-learning-procedural","slug":"differentiable-task-graph-learning-procedural","title":"Differentiable Task Graph Learning: Procedural Activity Representation and Online Mistake Detection from Egocentric Videos","date":"2024-06-03","arxiv_id":"2406.01486","repositories_listed":1,"syntology":null}],"record_sha256":"546195c9ff27a6e2e00a7d3461364a5be2caeb78b4578deeba5ed3e74de765be","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}