{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/6","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":12,"rows_per_page":100,"rows":[501,600],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding/papers/5","next":"/task/video-understanding/papers/7","papers":[{"url":"/paper/screencast-tutorial-video-understanding","slug":"screencast-tutorial-video-understanding","title":"Screencast Tutorial Video Understanding","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/carpe-posterum-a-convolutional-approach-for","slug":"carpe-posterum-a-convolutional-approach-for","title":"CARPe Posterum: A Convolutional Approach for Real-time Pedestrian Path Prediction","date":"2020-05-26","arxiv_id":"2005.12469","repositories_listed":1,"syntology":null},{"url":"/paper/dramaqa-character-centered-video-story","slug":"dramaqa-character-centered-video-story","title":"DramaQA: Character-Centered Video Story Understanding with Hierarchical QA","date":"2020-05-07","arxiv_id":"2005.03356","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-instructional-videos-probing-for-more","slug":"beyond-instructional-videos-probing-for-more","title":"Beyond Instructional Videos: Probing for More Diverse Visual-Textual Grounding on YouTube","date":"2020-04-29","arxiv_id":"2004.14338","repositories_listed":1,"syntology":null},{"url":"/paper/driftnet-aggressive-driving-behavior","slug":"driftnet-aggressive-driving-behavior","title":"DriftNet: Aggressive Driving Behavior Classification using 3D EfficientNet Architecture","date":"2020-04-18","arxiv_id":"2004.11970","repositories_listed":1,"syntology":null},{"url":"/paper/top-1-solution-of-multi-moments-in-time","slug":"top-1-solution-of-multi-moments-in-time","title":"Top-1 Solution of Multi-Moments in Time Challenge 2019","date":"2020-03-12","arxiv_id":"2003.05837","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-temporal-action-1","slug":"weakly-supervised-temporal-action-1","title":"Weakly Supervised Temporal Action Localization Using Deep Metric Learning","date":"2020-01-21","arxiv_id":"2001.07793","repositories_listed":1,"syntology":null},{"url":"/paper/tree-structured-policy-based-progressive","slug":"tree-structured-policy-based-progressive","title":"Tree-Structured Policy based Progressive Reinforcement Learning for Temporally Language Grounding in Video","date":"2020-01-18","arxiv_id":"2001.06680","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tree-structured-policy-based-progressive#ran","syntology_url":"https://syntology.ai/paper/2001.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06680"}},"official":{"repos":["WuJie1010/TSP-PRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/eev-dataset-predicting-expressions-evoked-by","slug":"eev-dataset-predicting-expressions-evoked-by","title":"EEV: A Large-Scale Dataset for Studying Evoked Expressions from Video","date":"2020-01-15","arxiv_id":"2001.05488","repositories_listed":1,"syntology":null},{"url":"/paper/comprehensive-soccer-video-understanding","slug":"comprehensive-soccer-video-understanding","title":"SoccerDB: A Large-Scale Database for Comprehensive Video Understanding","date":"2019-12-10","arxiv_id":"1912.04465","repositories_listed":1,"syntology":null},{"url":"/paper/stage-spatio-temporal-attention-on-graph","slug":"stage-spatio-temporal-attention-on-graph","title":"Video action detection by learning graph-based spatio-temporal interactions","date":"2019-12-09","arxiv_id":"1912.04316","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-pyramid-network-for-video-domain","slug":"adversarial-pyramid-network-for-video-domain","title":"VideoDG: Generalizing Temporal Relations in Videos to Novel Domains","date":"2019-12-08","arxiv_id":"1912.03716","repositories_listed":1,"syntology":null},{"url":"/paper/a-context-aware-loss-function-for-action","slug":"a-context-aware-loss-function-for-action","title":"A Context-Aware Loss Function for Action Spotting in Soccer Videos","date":"2019-12-03","arxiv_id":"1912.01326","repositories_listed":1,"syntology":null},{"url":"/paper/multi-attention-networks-for-temporal","slug":"multi-attention-networks-for-temporal","title":"Multi-attention Networks for Temporal Localization of Video-level Labels","date":"2019-11-15","arxiv_id":"1911.06866","repositories_listed":1,"syntology":null},{"url":"/paper/mod-a-deep-mixture-model-with-online","slug":"mod-a-deep-mixture-model-with-online","title":"MOD: A Deep Mixture Model with Online Knowledge Distillation for Large Scale Video Temporal Concept Localization","date":"2019-10-27","arxiv_id":"1910.12295","repositories_listed":1,"syntology":null},{"url":"/paper/vip-video-platform-for-pytorch","slug":"vip-video-platform-for-pytorch","title":"ViP: Video Platform for PyTorch","date":"2019-10-07","arxiv_id":"1910.02793","repositories_listed":1,"syntology":null},{"url":"/paper/gaussian-temporal-awareness-networks-for-1","slug":"gaussian-temporal-awareness-networks-for-1","title":"Gaussian Temporal Awareness Networks for Action Localization","date":"2019-09-09","arxiv_id":"1909.03877","repositories_listed":1,"syntology":null},{"url":"/paper/creative-flow-dataset","slug":"creative-flow-dataset","title":"Creative Flow+ Dataset","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/what-does-a-car-ssette-tape-tell","slug":"what-does-a-car-ssette-tape-tell","title":"Audio Caption in a Car Setting with a Sentence-Level Loss","date":"2019-05-31","arxiv_id":"1905.13448","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-temporal-information-for-improved","slug":"exploring-temporal-information-for-improved","title":"Exploring Temporal Information for Improved Video Understanding","date":"2019-05-25","arxiv_id":"1905.10654","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-network-architecture-for-real","slug":"lightweight-network-architecture-for-real","title":"Lightweight Network Architecture for Real-Time Action Recognition","date":"2019-05-21","arxiv_id":"1905.08711","repositories_listed":1,"syntology":null},{"url":"/paper/holistic-large-scale-video-understanding","slug":"holistic-large-scale-video-understanding","title":"Large Scale Holistic Video Understanding","date":"2019-04-25","arxiv_id":"1904.11451","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-space-time-graphs-for-video","slug":"recurrent-space-time-graphs-for-video","title":"Recurrent Space-time Graph Neural Networks","date":"2019-04-11","arxiv_id":"1904.05582","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-space-time-graphs-for-video#ran","syntology_url":"https://syntology.ai/paper/1904.05582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.05582"}},"official":{"repos":["IuliaDuta/RSTG"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/4d-generic-video-object-proposals","slug":"4d-generic-video-object-proposals","title":"4D Generic Video Object Proposals","date":"2019-01-26","arxiv_id":"1901.09260","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/4d-generic-video-object-proposals#ran","syntology_url":"https://syntology.ai/paper/1901.09260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09260"}},"official":{"repos":["aljosaosep/4DGVT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-visual-centrifuge-model-free-layered","slug":"the-visual-centrifuge-model-free-layered","title":"The Visual Centrifuge: Model-Free Layered Video Representations","date":"2018-12-04","arxiv_id":"1812.01461","repositories_listed":1,"syntology":null},{"url":"/paper/nextvlad-an-efficient-neural-network-to","slug":"nextvlad-an-efficient-neural-network-to","title":"NeXtVLAD: An Efficient Neural Network to Aggregate Frame-level Features for Large-scale Video Classification","date":"2018-11-12","arxiv_id":"1811.05014","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/nextvlad-an-efficient-neural-network-to#ran","syntology_url":"https://syntology.ai/paper/1811.05014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.05014"}},"official":null}},{"url":"/paper/unsupervised-adversarial-visual-level-domain","slug":"unsupervised-adversarial-visual-level-domain","title":"Unsupervised Adversarial Visual Level Domain Adaptation for Learning Video Object Detectors from Images","date":"2018-10-04","arxiv_id":"1810.02074","repositories_listed":1,"syntology":null},{"url":"/paper/learnable-pooling-methods-for-video","slug":"learnable-pooling-methods-for-video","title":"Learnable Pooling Methods for Video Classification","date":"2018-10-01","arxiv_id":"1810.00530","repositories_listed":1,"syntology":null},{"url":"/paper/localizing-moments-in-video-with-temporal","slug":"localizing-moments-in-video-with-temporal","title":"Localizing Moments in Video with Temporal Language","date":"2018-09-05","arxiv_id":"1809.01337","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-error-in-temporal-action-detectors","slug":"diagnosing-error-in-temporal-action-detectors","title":"Diagnosing Error in Temporal Action Detectors","date":"2018-07-27","arxiv_id":"1807.10706","repositories_listed":1,"syntology":null},{"url":"/paper/watch-listen-and-describe-globally-and","slug":"watch-listen-and-describe-globally-and","title":"Watch, Listen, and Describe: Globally and Locally Aligned Cross-Modal Attentions for Video Captioning","date":"2018-04-15","arxiv_id":"1804.05448","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-motion-representation","slug":"end-to-end-learning-of-motion-representation","title":"End-to-End Learning of Motion Representation for Video Understanding","date":"2018-04-02","arxiv_id":"1804.00413","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/end-to-end-learning-of-motion-representation#ran","syntology_url":"https://syntology.ai/paper/1804.00413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00413"}},"official":null}},{"url":"/paper/joint-event-detection-and-description-in","slug":"joint-event-detection-and-description-in","title":"Joint Event Detection and Description in Continuous Video Streams","date":"2018-02-28","arxiv_id":"1802.10250","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/joint-event-detection-and-description-in#ran","syntology_url":"https://syntology.ai/paper/1802.10250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10250"}},"official":{"repos":["VisionLearningGroup/JEDDi-Net"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/detect-and-track-efficient-pose-estimation-in","slug":"detect-and-track-efficient-pose-estimation-in","title":"Detect-and-Track: Efficient Pose Estimation in Videos","date":"2017-12-26","arxiv_id":"1712.09184","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-modeling-approaches-for-large-scale","slug":"temporal-modeling-approaches-for-large-scale","title":"Temporal Modeling Approaches for Large-scale Youtube-8M Video Understanding","date":"2017-07-14","arxiv_id":"1707.04555","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-deep-recurrent-architecture-for","slug":"hierarchical-deep-recurrent-architecture-for","title":"Hierarchical Deep Recurrent Architecture for Video Understanding","date":"2017-07-11","arxiv_id":"1707.03296","repositories_listed":1,"syntology":null},{"url":"/paper/the-youtube-8m-kaggle-competition-challenges","slug":"the-youtube-8m-kaggle-competition-challenges","title":"The YouTube-8M Kaggle Competition: Challenges and Methods","date":"2017-06-28","arxiv_id":"1706.09274","repositories_listed":1,"syntology":null},{"url":"/paper/the-monkeytyping-solution-to-the-youtube-8m","slug":"the-monkeytyping-solution-to-the-youtube-8m","title":"The Monkeytyping Solution to the YouTube-8M Video Understanding Challenge","date":"2017-06-16","arxiv_id":"1706.05150","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-methods-for-efficient-large","slug":"deep-learning-methods-for-efficient-large","title":"Deep Learning Methods for Efficient Large Scale Video Labeling","date":"2017-06-14","arxiv_id":"1706.04572","repositories_listed":1,"syntology":null},{"url":"/paper/video-object-segmentation-using-supervoxel","slug":"video-object-segmentation-using-supervoxel","title":"Video Object Segmentation using Supervoxel-Based Gerrymandering","date":"2017-04-18","arxiv_id":"1704.05165","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-tessellation-a-unified-approach-for","slug":"temporal-tessellation-a-unified-approach-for","title":"Temporal Tessellation: A Unified Approach for Video Analysis","date":"2016-12-21","arxiv_id":"1612.06950","repositories_listed":1,"syntology":null},{"url":"/paper/pooled-motion-features-for-first-person","slug":"pooled-motion-features-for-first-person","title":"Pooled Motion Features for First-Person Videos","date":"2014-12-19","arxiv_id":"1412.6505","repositories_listed":1,"syntology":null},{"url":null,"slug":"videoitg-multimodal-video-understanding-with","title":"VideoITG: Multimodal Video Understanding with Instructed Temporal Grounding","date":"2025-07-17","arxiv_id":"2507.13353","repositories_listed":0,"syntology":null},{"url":null,"slug":"chat-with-ai-the-surprising-turn-of-real-time","title":"Chat with AI: The Surprising Turn of Real-time Video Communication from Human to AI","date":"2025-07-14","arxiv_id":"2507.10510","repositories_listed":0,"syntology":null},{"url":null,"slug":"embrace-3k-embodied-reasoning-and-action-in","title":"EmbRACE-3K: Embodied Reasoning and Action in Complex Environments","date":"2025-07-14","arxiv_id":"2507.10548","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-appearance-geometric-cues-for-robust","title":"Beyond Appearance: Geometric Cues for Robust Video Instance Segmentation","date":"2025-07-08","arxiv_id":"2507.05948","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-event-reasoning-and-prediction-by","title":"Video Event Reasoning and Prediction by Fusing World Knowledge from LLMs with Vision Foundation Models","date":"2025-07-08","arxiv_id":"2507.05822","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-crash-detection-in","title":"Large Language Models for Crash Detection in Video: A Survey of Methods, Datasets, and Challenges","date":"2025-07-02","arxiv_id":"2507.02074","repositories_listed":0,"syntology":null},{"url":null,"slug":"cavalry-v-a-large-scale-generator-framework","title":"CAVALRY-V: A Large-Scale Generator Framework for Adversarial Attacks on Video MLLMs","date":"2025-07-01","arxiv_id":"2507.00817","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-frame-query-aware-frame-selection-and-multi","title":"Q-Frame: Query-aware Frame Selection and Multi-Resolution Adaptation for Video-LLMs","date":"2025-06-27","arxiv_id":"2506.22139","repositories_listed":0,"syntology":null},{"url":null,"slug":"ipformer-videollm-enhancing-multi-modal-video","title":"IPFormer-VideoLLM: Enhancing Multi-modal Video Understanding for Multi-shot Scenes","date":"2025-06-26","arxiv_id":"2506.21116","repositories_listed":0,"syntology":null},{"url":null,"slug":"pevlm-parallel-encoding-for-vision-language","title":"PEVLM: Parallel Encoding for Vision-Language Models","date":"2025-06-24","arxiv_id":"2506.19651","repositories_listed":0,"syntology":null},{"url":null,"slug":"grpo-care-consistency-aware-reinforcement","title":"GRPO-CARE: Consistency-Aware Reinforcement Learning for Multimodal Reasoning","date":"2025-06-19","arxiv_id":"2506.16141","repositories_listed":0,"syntology":null},{"url":null,"slug":"infinipot-v-memory-constrained-kv-cache","title":"InfiniPot-V: Memory-Constrained KV Cache Compression for Streaming Video Understanding","date":"2025-06-18","arxiv_id":"2506.15745","repositories_listed":0,"syntology":null},{"url":null,"slug":"adavideorag-omni-contextual-adaptive","title":"AdaVideoRAG: Omni-Contextual Adaptive Retrieval-Augmented Efficient Long Video Understanding","date":"2025-06-16","arxiv_id":"2506.13589","repositories_listed":0,"syntology":null},{"url":null,"slug":"mambamia-a-state-space-model-based","title":"MambaMia: A State-Space-Model-Based Compression for Efficient Video Understanding in Large Multimodal Models","date":"2025-06-16","arxiv_id":"2506.13564","repositories_listed":0,"syntology":null},{"url":null,"slug":"mlvtg-mamba-based-feature-alignment-and-llm","title":"MLVTG: Mamba-Based Feature Alignment and LLM-Driven Purification for Multi-Modal Video Temporal Grounding","date":"2025-06-10","arxiv_id":"2506.08512","repositories_listed":0,"syntology":null},{"url":null,"slug":"versavid-r1-a-versatile-video-understanding","title":"VersaVid-R1: A Versatile Video Understanding and Reasoning Model from Question Answering to Captioning Tasks","date":"2025-06-10","arxiv_id":"2506.09079","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenerag-scene-level-retrieval-augmented","title":"SceneRAG: Scene-level Retrieval-Augmented Generation for Video Understanding","date":"2025-06-09","arxiv_id":"2506.07600","repositories_listed":0,"syntology":null},{"url":null,"slug":"super-encoding-network-recursive-association","title":"Super Encoding Network: Recursive Association of Multi-Modal Encoders for Video Understanding","date":"2025-06-09","arxiv_id":"2506.07576","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgbench-a-unified-large-scale-benchmark-for","title":"SurgBench: A Unified Large-Scale Benchmark for Surgical Video Analysis","date":"2025-06-09","arxiv_id":"2506.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-audio-and-vision-zero-shot","title":"Bridging Audio and Vision: Zero-Shot Audiovisual Segmentation by Connecting Pretrained Models","date":"2025-06-06","arxiv_id":"2506.06537","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-perspectives-a-survey-on-cross-view","title":"Bridging Perspectives: A Survey on Cross-view Collaborative Intelligence with Egocentric-Exocentric Vision","date":"2025-06-06","arxiv_id":"2506.06253","repositories_listed":0,"syntology":null},{"url":null,"slug":"apvr-hour-level-long-video-understanding-with","title":"APVR: Hour-Level Long Video Understanding with Adaptive Pivot Visual Information Retrieval","date":"2025-06-05","arxiv_id":"2506.04953","repositories_listed":0,"syntology":null},{"url":null,"slug":"av-reasoner-improving-and-benchmarking-clue","title":"AV-Reasoner: Improving and Benchmarking Clue-Grounded Audio-Visual Counting for MLLMs","date":"2025-06-05","arxiv_id":"2506.05328","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualx-vsr-dual-axial-spatial-times-temporal","title":"DualX-VSR: Dual Axial Spatial$\\times$Temporal Transformer for Real-World Video Super-Resolution without Motion Compensation","date":"2025-06-05","arxiv_id":"2506.04830","repositories_listed":0,"syntology":null},{"url":null,"slug":"textvidbench-a-benchmark-for-long-video-scene","title":"TextVidBench: A Benchmark for Long Video Scene Text Understanding","date":"2025-06-05","arxiv_id":"2506.04983","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyntok-dynamic-compression-of-visual-tokens","title":"DynTok: Dynamic Compression of Visual Tokens for Efficient and Effective Video Understanding","date":"2025-06-04","arxiv_id":"2506.03990","repositories_listed":0,"syntology":null},{"url":null,"slug":"interrvos-interaction-aware-referring-video","title":"InterRVOS: Interaction-aware Referring Video Object Segmentation","date":"2025-06-03","arxiv_id":"2506.02356","repositories_listed":0,"syntology":null},{"url":null,"slug":"reagent-v-a-reward-driven-multi-agent","title":"ReAgent-V: A Reward-Driven Multi-Agent Framework for Video Understanding","date":"2025-06-02","arxiv_id":"2506.01300","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexselect-flexible-token-selection-for","title":"FlexSelect: Flexible Token Selection for Efficient Long Video Understanding","date":"2025-06-01","arxiv_id":"2506.00993","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-detection-policies-and-keyframe","title":"Scene Detection Policies and Keyframe Extraction Strategies for Large-Scale Video Analysis","date":"2025-05-31","arxiv_id":"2506.00667","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-reusable-concepts-across-different","title":"Learning reusable concepts across different egocentric video understanding tasks","date":"2025-05-30","arxiv_id":"2505.24690","repositories_listed":0,"syntology":null},{"url":null,"slug":"threading-keyframe-with-narratives-mllms-as","title":"Threading Keyframe with Narratives: MLLMs as Strong Long Video Comprehenders","date":"2025-05-30","arxiv_id":"2505.24158","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-blindness-why-video-language-models-can","title":"Time Blindness: Why Video-Language Models Can't See What Humans Can?","date":"2025-05-30","arxiv_id":"2505.24867","repositories_listed":0,"syntology":null},{"url":null,"slug":"vudg-a-dataset-for-video-understanding-domain","title":"VUDG: A Dataset for Video Understanding Domain Generalization","date":"2025-05-30","arxiv_id":"2505.24346","repositories_listed":0,"syntology":null},{"url":null,"slug":"macp-minimal-yet-mighty-adaptation-via","title":"MaCP: Minimal yet Mighty Adaptation via Hierarchical Cosine Projection","date":"2025-05-29","arxiv_id":"2505.23870","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-rag-a-multimodal-retrieval-augmented","title":"Multi-RAG: A Multimodal Retrieval-Augmented Generation System for Adaptive Video Understanding","date":"2025-05-29","arxiv_id":"2505.23990","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-visuo-tactile-video-understanding","title":"Universal Visuo-Tactile Video Understanding for Embodied Interaction","date":"2025-05-28","arxiv_id":"2505.22566","repositories_listed":0,"syntology":null},{"url":null,"slug":"adatp-attention-debiased-token-pruning-for","title":"AdaTP: Attention-Debiased Token Pruning for Video Large Language Models","date":"2025-05-26","arxiv_id":"2505.20100","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-causally-related-needles-in-a-video","title":"Two Causally Related Needles in a Video Haystack","date":"2025-05-26","arxiv_id":"2505.19853","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-to-dense-a-free-lunch-for-lossless","title":"Sparse-to-Dense: A Free Lunch for Lossless Acceleration of Video Understanding in LLMs","date":"2025-05-25","arxiv_id":"2505.19155","repositories_listed":0,"syntology":null},{"url":"/paper/deep-video-discovery-agentic-search-with-tool","slug":"deep-video-discovery-agentic-search-with-tool","title":"Deep Video Discovery: Agentic Search with Tool Use for Long-form Video Understanding","date":"2025-05-23","arxiv_id":"2505.18079","repositories_listed":0,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-video-discovery-agentic-search-with-tool#ran","syntology_url":"https://syntology.ai/paper/2505.18079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18079"}},"official":null}},{"url":null,"slug":"four-eyes-are-better-than-two-harnessing-the","title":"Four Eyes Are Better Than Two: Harnessing the Collaborative Potential of Large Models via Differentiated Thinking and Complementary Ensembles","date":"2025-05-22","arxiv_id":"2505.16784","repositories_listed":0,"syntology":null},{"url":null,"slug":"soccerchat-integrating-multimodal-data-for","title":"SoccerChat: Integrating Multimodal Data for Enhanced Soccer Game Understanding","date":"2025-05-22","arxiv_id":"2505.16630","repositories_listed":0,"syntology":null},{"url":null,"slug":"clapper-compact-learning-and-video","title":"Clapper: Compact Learning and Video Representation in VLMs","date":"2025-05-21","arxiv_id":"2505.15529","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-foundation-models-for-multimodal","title":"Leveraging Foundation Models for Multimodal Graph-Based Action Recognition","date":"2025-05-21","arxiv_id":"2505.15192","repositories_listed":0,"syntology":null},{"url":null,"slug":"livevlm-efficient-online-video-understanding","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","date":"2025-05-21","arxiv_id":"2505.15269","repositories_listed":0,"syntology":null},{"url":null,"slug":"viarl-adaptive-temporal-grounding-via-visual","title":"ViaRL: Adaptive Temporal Grounding via Visual Iterated Amplification Reinforcement Learning","date":"2025-05-21","arxiv_id":"2505.15447","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-down-video-llm-benchmarks-knowledge","title":"Breaking Down Video LLM Benchmarks: Knowledge, Spatial Perception, or True Temporal Understanding?","date":"2025-05-20","arxiv_id":"2505.14321","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-of-vlm-for-soccer-video","title":"Domain Adaptation of VLM for Soccer Video Understanding","date":"2025-05-20","arxiv_id":"2505.13860","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-shots-to-stories-llm-assisted-video","title":"From Shots to Stories: LLM-Assisted Video Editing with Unified Language Representations","date":"2025-05-18","arxiv_id":"2505.12237","repositories_listed":0,"syntology":null},{"url":null,"slug":"skillformer-unified-multi-view-video","title":"SkillFormer: Unified Multi-View Video Understanding for Proficiency Estimation","date":"2025-05-13","arxiv_id":"2505.08665","repositories_listed":0,"syntology":null},{"url":null,"slug":"gameplay-highlights-generation","title":"Gameplay Highlights Generation","date":"2025-05-12","arxiv_id":"2505.07721","repositories_listed":0,"syntology":null},{"url":"/paper/seed1-5-vl-technical-report","slug":"seed1-5-vl-technical-report","title":"Seed1.5-VL Technical Report","date":"2025-05-11","arxiv_id":"2505.07062","repositories_listed":0,"syntology":null},{"url":null,"slug":"streambridge-turning-your-offline-video-large","title":"StreamBridge: Turning Your Offline Video Large Language Model into a Proactive Streaming Assistant","date":"2025-05-08","arxiv_id":"2505.05467","repositories_listed":0,"syntology":null},{"url":null,"slug":"ravu-retrieval-augmented-video-understanding","title":"RAVU: Retrieval Augmented Video Understanding with Compositional Reasoning over Graph","date":"2025-05-06","arxiv_id":"2505.03173","repositories_listed":0,"syntology":null},{"url":null,"slug":"videollm-benchmarks-and-evaluation-a-survey","title":"VideoLLM Benchmarks and Evaluation: A Survey","date":"2025-05-03","arxiv_id":"2505.03829","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-agentic-video-analytics-systems","title":"Empowering Agentic Video Analytics Systems with Video Language Models","date":"2025-05-01","arxiv_id":"2505.00254","repositories_listed":0,"syntology":null},{"url":null,"slug":"timesoccer-an-end-to-end-multimodal-large","title":"TimeSoccer: An End-to-End Multimodal Large Language Model for Soccer Commentary Generation","date":"2025-04-24","arxiv_id":"2504.17365","repositories_listed":0,"syntology":null}],"record_sha256":"b4d5550992c26f5e2877fc8e3e5713630ee1283f8a32ef6326301a2e0215e6fb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}