{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/7","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":12,"rows_per_page":100,"rows":[601,700],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding/papers/6","next":"/task/video-understanding/papers/8","papers":[{"url":null,"slug":"dymu-dynamic-merging-and-virtual-unmerging","title":"DyMU: Dynamic Merging and Virtual Unmerging for Efficient VLMs","date":"2025-04-23","arxiv_id":"2504.17040","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-lmm-for-efficient-video-understanding-via","title":"An LMM for Efficient Video Understanding via Reinforced Compression of Video Cubes","date":"2025-04-21","arxiv_id":"2504.15270","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-md-grounded-video-language-pre","title":"Grounding-MD: Grounded Video-language Pre-training for Open-World Moment Detection","date":"2025-04-20","arxiv_id":"2504.14553","repositories_listed":0,"syntology":null},{"url":null,"slug":"omniv-med-scaling-medical-vision-language","title":"OmniV-Med: Scaling Medical Vision-Language Model for Universal Visual Understanding","date":"2025-04-20","arxiv_id":"2504.14692","repositories_listed":0,"syntology":null},{"url":null,"slug":"resnetvllm-multi-modal-vision-llm-for-the","title":"ResNetVLLM -- Multi-modal Vision LLM for the Video Understanding Task","date":"2025-04-20","arxiv_id":"2504.14432","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-well-can-general-vision-language-models","title":"How Well Can General Vision-Language Models Learn Medicine By Watching Public Educational Videos?","date":"2025-04-19","arxiv_id":"2504.14391","repositories_listed":0,"syntology":null},{"url":null,"slug":"prototypes-are-balanced-units-for-efficient","title":"Prototypes are Balanced Units for Efficient and Effective Partially Relevant Video Retrieval","date":"2025-04-17","arxiv_id":"2504.13035","repositories_listed":0,"syntology":null},{"url":"/paper/self-alignment-of-large-video-language-models","slug":"self-alignment-of-large-video-language-models","title":"Self-alignment of Large Video Language Models with Refined Regularized Preference Optimization","date":"2025-04-16","arxiv_id":"2504.12083","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnivdiff-omni-controllable-video-diffusion","title":"OmniVDiff: Omni Controllable Video Diffusion for Generation and Understanding","date":"2025-04-15","arxiv_id":"2504.10825","repositories_listed":0,"syntology":null},{"url":null,"slug":"pvuw-2025-challenge-report-advances-in-pixel","title":"PVUW 2025 Challenge Report: Advances in Pixel-level Understanding of Complex Videos in the Wild","date":"2025-04-15","arxiv_id":"2504.11326","repositories_listed":0,"syntology":null},{"url":null,"slug":"mavors-multi-granularity-video-representation","title":"Mavors: Multi-granularity Video Representation for Multimodal Large Language Model","date":"2025-04-14","arxiv_id":"2504.10068","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-and-robust-moment-retrieval","title":"Towards Efficient and Robust Moment Retrieval System: A Unified Framework for Multi-Granularity Models and Temporal Reranking","date":"2025-04-11","arxiv_id":"2504.08384","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-can-objects-help-video-language","title":"How Can Objects Help Video-Language Understanding?","date":"2025-04-10","arxiv_id":"2504.07454","repositories_listed":0,"syntology":null},{"url":null,"slug":"sf2t-self-supervised-fragment-finetuning-of","title":"SF2T: Self-supervised Fragment Finetuning of Video-LLMs for Fine-Grained Understanding","date":"2025-04-10","arxiv_id":"2504.07745","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoexpert-augmented-llm-for-temporal","title":"VideoExpert: Augmented LLM for Temporal-Sensitive Video Understanding","date":"2025-04-10","arxiv_id":"2504.07519","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-128k-to-4m-efficient-training-of-ultra","title":"From 128K to 4M: Efficient Training of Ultra-Long Context Large Language Models","date":"2025-04-08","arxiv_id":"2504.06214","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-broadcast-to-minimap-achieving-state-of","title":"From Broadcast to Minimap: Achieving State-of-the-Art SoccerNet Game State Reconstruction","date":"2025-04-08","arxiv_id":"2504.06357","repositories_listed":0,"syntology":null},{"url":null,"slug":"instructionbench-an-instructional-video","title":"InstructionBench: An Instructional Video Understanding Benchmark","date":"2025-04-07","arxiv_id":"2504.05040","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-audio-guided-video-representation","title":"Learning Audio-guided Video Representation with Gated Attention for Video-Text Retrieval","date":"2025-04-03","arxiv_id":"2504.02397","repositories_listed":0,"syntology":null},{"url":null,"slug":"moment-quantization-for-video-temporal","title":"Moment Quantization for Video Temporal Grounding","date":"2025-04-03","arxiv_id":"2504.02286","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligned-better-listen-better-for-audio-visual","title":"Aligned Better, Listen Better for Audio-Visual Large Language Models","date":"2025-04-02","arxiv_id":"2504.02061","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-temporal-prompting-all-we-need-for-limited","title":"Is Temporal Prompting All We Need For Limited Labeled Action Recognition?","date":"2025-04-02","arxiv_id":"2504.01890","repositories_listed":0,"syntology":null},{"url":null,"slug":"timesearch-hierarchical-video-search-with","title":"TimeSearch: Hierarchical Video Search with Spotlight and Reflection for Human-like Long Video Understanding","date":"2025-04-02","arxiv_id":"2504.01407","repositories_listed":0,"syntology":null},{"url":null,"slug":"dante-ad-dual-vision-attention-network-for","title":"DANTE-AD: Dual-Vision Attention Network for Long-Term Audio Description","date":"2025-03-31","arxiv_id":"2503.24096","repositories_listed":0,"syntology":null},{"url":null,"slug":"h2vu-benchmark-a-comprehensive-benchmark-for","title":"H2VU-Benchmark: A Comprehensive Benchmark for Hierarchical Holistic Video Understanding","date":"2025-03-31","arxiv_id":"2503.24008","repositories_listed":0,"syntology":null},{"url":"/paper/ca-2st-cross-attention-in-audio-space-and","slug":"ca-2st-cross-attention-in-audio-space-and","title":"CA^2ST: Cross-Attention in Audio, Space, and Time for Holistic Video Recognition","date":"2025-03-30","arxiv_id":"2503.23447","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnimmi-a-comprehensive-multi-modal","title":"OmniMMI: A Comprehensive Multi-modal Interaction Benchmark in Streaming Video Contexts","date":"2025-03-29","arxiv_id":"2503.22952","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-trial-to-triumph-advancing-long-video","title":"From Trial to Triumph: Advancing Long Video Understanding via Visual Context Sample Scaling and Self-reward Alignment","date":"2025-03-26","arxiv_id":"2503.20472","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-res-self-reflection-in-large-vision","title":"Self-ReS: Self-Reflection in Large Vision-Language Models for Long Video Understanding","date":"2025-03-26","arxiv_id":"2503.20362","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-encoder-barrier-for-seamless","title":"Breaking the Encoder Barrier for Seamless Video-Language Understanding","date":"2025-03-24","arxiv_id":"2503.18422","repositories_listed":0,"syntology":null},{"url":null,"slug":"crcl-causal-representation-consistency","title":"CRCL: Causal Representation Consistency Learning for Anomaly Detection in Surveillance Videos","date":"2025-03-24","arxiv_id":"2503.18808","repositories_listed":0,"syntology":null},{"url":null,"slug":"slowfast-llava-1-5-a-family-of-token","title":"SlowFast-LLaVA-1.5: A Family of Token-Efficient Video Large Language Models for Long-Form Video Understanding","date":"2025-03-24","arxiv_id":"2503.18943","repositories_listed":0,"syntology":null},{"url":null,"slug":"unbiasing-through-textual-descriptions","title":"Unbiasing through Textual Descriptions: Mitigating Representation Bias in Video Benchmarks","date":"2025-03-24","arxiv_id":"2503.18637","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-xl-pro-reconstructive-token-compression","title":"Video-XL-Pro: Reconstructive Token Compression for Extremely Long Video Understanding","date":"2025-03-24","arxiv_id":"2503.18478","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-temporal-consistency-learning","title":"Collaborative Temporal Consistency Learning for Point-supervised Natural Language Video Localization","date":"2025-03-22","arxiv_id":"2503.17651","repositories_listed":0,"syntology":null},{"url":null,"slug":"pvchat-personalized-video-chat-with-one-shot","title":"PVChat: Personalized Video Chat with One-Shot Learning","date":"2025-03-21","arxiv_id":"2503.17069","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-action-detection-model-compression","title":"Temporal Action Detection Model Compression by Progressive Block Drop","date":"2025-03-21","arxiv_id":"2503.16916","repositories_listed":0,"syntology":null},{"url":null,"slug":"docvideoqa-towards-comprehensive","title":"DocVideoQA: Towards Comprehensive Understanding of Document-Centric Videos through Question Answering","date":"2025-03-20","arxiv_id":"2503.15887","repositories_listed":0,"syntology":null},{"url":null,"slug":"mash-vlm-mitigating-action-scene","title":"MASH-VLM: Mitigating Action-Scene Hallucination in Video-LLMs through Disentangled Spatial-Temporal Representations","date":"2025-03-20","arxiv_id":"2503.15871","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-can-off-the-shelves-large-multi-modal","title":"What can Off-the-Shelves Large Multi-Modal Models do for Dynamic Scene Graph Generation?","date":"2025-03-20","arxiv_id":"2503.15846","repositories_listed":0,"syntology":null},{"url":null,"slug":"favor-bench-a-comprehensive-benchmark-for","title":"FAVOR-Bench: A Comprehensive Benchmark for Fine-Grained Video Motion Understanding","date":"2025-03-19","arxiv_id":"2503.14935","repositories_listed":0,"syntology":null},{"url":null,"slug":"impossible-videos","title":"Impossible Videos","date":"2025-03-18","arxiv_id":"2503.14378","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-llm-video-understanding-with-16","title":"Improving LLM Video Understanding with 16 Frames Per Second","date":"2025-03-18","arxiv_id":"2503.13956","repositories_listed":0,"syntology":null},{"url":null,"slug":"spacevllm-endowing-multimodal-large-language","title":"SpaceVLLM: Endowing Multimodal Large Language Model with Spatio-Temporal Video Grounding Capability","date":"2025-03-18","arxiv_id":"2503.13983","repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-in-frames-dynamic-keyframe-search-via","title":"Logic-in-Frames: Dynamic Keyframe Search via Visual Semantic-Logical Verification for Long Video Understanding","date":"2025-03-17","arxiv_id":"2503.13139","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-vmnet-accelerating-long-form-video","title":"Long-VMNet: Accelerating Long-Form Video Understanding via Fixed Memory","date":"2025-03-17","arxiv_id":"2503.13707","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-scalable-modeling-of-compressed","title":"Towards Scalable Modeling of Compressed Videos for Efficient Action Recognition","date":"2025-03-17","arxiv_id":"2503.13724","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-mlb-mitigating-and-leveraging-attention","title":"LLaVA-MLB: Mitigating and Leveraging Attention Bias for Training-Free Video LLMs","date":"2025-03-14","arxiv_id":"2503.11205","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-star-benchmarking-video-llms-on-video","title":"V-STaR: Benchmarking Video-LLMs on Video Spatio-Temporal Reasoning","date":"2025-03-14","arxiv_id":"2503.11495","repositories_listed":0,"syntology":null},{"url":null,"slug":"vamba-understanding-hour-long-videos-with","title":"Vamba: Understanding Hour-Long Videos with Hybrid Mamba-Transformers","date":"2025-03-14","arxiv_id":"2503.11579","repositories_listed":0,"syntology":null},{"url":null,"slug":"watch-and-learn-leveraging-expert-knowledge","title":"Watch and Learn: Leveraging Expert Knowledge and Language for Surgical Video Understanding","date":"2025-03-14","arxiv_id":"2503.11392","repositories_listed":0,"syntology":null},{"url":"/paper/lvagent-long-video-understanding-by-multi","slug":"lvagent-long-video-understanding-by-multi","title":"LVAgent: Long Video Understanding by Multi-Round Dynamical Collaboration of MLLM Agents","date":"2025-03-13","arxiv_id":"2503.10200","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvagent-long-video-understanding-by-multi#ran","syntology_url":"https://syntology.ai/paper/2503.10200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10200"}},"official":null}},{"url":null,"slug":"time-temporal-sensitive-multi-dimensional","title":"TIME: Temporal-sensitive Multi-dimensional Instruction Tuning and Benchmarking for Video-LLMs","date":"2025-03-13","arxiv_id":"2503.09994","repositories_listed":0,"syntology":null},{"url":null,"slug":"everything-can-be-described-in-words-a-simple","title":"Everything Can Be Described in Words: A Simple Unified Multi-Modal Framework with Semantic and Temporal Alignment","date":"2025-03-12","arxiv_id":"2503.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"exo2ego-exocentric-knowledge-guided-mllm-for","title":"Exo2Ego: Exocentric Knowledge Guided MLLM for Egocentric Video Understanding","date":"2025-03-12","arxiv_id":"2503.09143","repositories_listed":0,"syntology":null},{"url":null,"slug":"favchat-unlocking-fine-grained-facail-video","title":"FaVChat: Unlocking Fine-Grained Facail Video Understanding with Multimodal Large Language Models","date":"2025-03-12","arxiv_id":"2503.09158","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-frame-sampler-for-long-video","title":"Generative Frame Sampler for Long Video Understanding","date":"2025-03-12","arxiv_id":"2503.09146","repositories_listed":0,"syntology":null},{"url":null,"slug":"measure-twice-cut-once-grasping-video","title":"Measure Twice, Cut Once: Grasping Video Structures and Event Semantics with LLMs for Video Temporal Localization","date":"2025-03-12","arxiv_id":"2503.09027","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-enhanced-retrieval-augmentation-for","title":"Memory-enhanced Retrieval Augmentation for Long Video Understanding","date":"2025-03-12","arxiv_id":"2503.09149","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-limitations-of-vision-language-models","title":"On the Limitations of Vision-Language Models in Understanding Image Transforms","date":"2025-03-12","arxiv_id":"2503.09837","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-is-all-you-need-for-video","title":"Reasoning is All You Need for Video Generalization: A Counterfactual Benchmark with Sub-question Evaluation","date":"2025-03-12","arxiv_id":"2503.10691","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoscan-enabling-efficient-streaming-video","title":"VideoScan: Enabling Efficient Streaming Video Understanding via Frame-level Semantic Carriers","date":"2025-03-12","arxiv_id":"2503.09387","repositories_listed":0,"syntology":null},{"url":null,"slug":"allvb-all-in-one-long-video-understanding","title":"ALLVB: All-in-One Long Video Understanding Benchmark","date":"2025-03-10","arxiv_id":"2503.07298","repositories_listed":0,"syntology":null},{"url":null,"slug":"bearcubs-a-benchmark-for-computer-using-web","title":"BEARCUBS: A benchmark for computer-using web agents","date":"2025-03-10","arxiv_id":"2503.07919","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-fine-grained-video-question-answering","title":"Towards Fine-Grained Video Question Answering","date":"2025-03-10","arxiv_id":"2503.06820","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-visual-discrimination-and-reasoning","title":"Towards Visual Discrimination and Reasoning of Real-World Physical Dynamics: Physics-Grounded Anomaly Detection","date":"2025-03-05","arxiv_id":"2503.03562","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-00162","title":"PreMind: Multi-Agent Video Understanding for Advanced Indexing of Presentation-style Videos","date":"2025-02-28","arxiv_id":"2503.00162","repositories_listed":0,"syntology":null},{"url":null,"slug":"haic-improving-human-action-understanding-and","title":"HAIC: Improving Human Action Understanding and Generation with Better Captions for Multi-modal Large Language Models","date":"2025-02-28","arxiv_id":"2502.20811","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-llm-based-video-frame-selection-for","title":"M-LLM Based Video Frame Selection for Efficient Video Understanding","date":"2025-02-27","arxiv_id":"2502.19680","repositories_listed":0,"syntology":null},{"url":null,"slug":"internvqa-advancing-compressed-video-quality","title":"InternVQA: Advancing Compressed Video Quality Assessment with Distilling Large Foundation Model","date":"2025-02-26","arxiv_id":"2502.19026","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-00042","title":"An Analysis of Data Transformation Effects on Segment Anything 2","date":"2025-02-25","arxiv_id":"2503.00042","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-video-captioning-through-scene","title":"Fine-Grained Video Captioning through Scene Graph Consolidation","date":"2025-02-23","arxiv_id":"2502.16427","repositories_listed":0,"syntology":null},{"url":null,"slug":"longcaptioning-unlocking-the-power-of-long","title":"LongCaptioning: Unlocking the Power of Long Caption Generation in Large Multimodal Models","date":"2025-02-21","arxiv_id":"2502.15393","repositories_listed":0,"syntology":null},{"url":null,"slug":"avd2-accident-video-diffusion-for-accident","title":"AVD2: Accident Video Diffusion for Accident Video Description","date":"2025-02-20","arxiv_id":"2502.14801","repositories_listed":0,"syntology":null},{"url":null,"slug":"momentseeker-a-comprehensive-benchmark-and-a","title":"MomentSeeker: A Task-Oriented Benchmark For Long-Video Moment Retrieval","date":"2025-02-18","arxiv_id":"2502.12558","repositories_listed":0,"syntology":null},{"url":null,"slug":"imove-instance-motion-aware-video","title":"iMOVE: Instance-Motion-Aware Video Understanding","date":"2025-02-17","arxiv_id":"2502.11594","repositories_listed":0,"syntology":null},{"url":"/paper/semantics-aware-test-time-adaptation-for-3d","slug":"semantics-aware-test-time-adaptation-for-3d","title":"Semantics-aware Test-time Adaptation for 3D Human Pose Estimation","date":"2025-02-15","arxiv_id":"2502.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-gpt-for-video-understanding-zero","title":"Optimizing GPT for Video Understanding: Zero-Shot Performance and Prompt Engineering","date":"2025-02-13","arxiv_id":"2502.09573","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-mamba-architecture-for-vision","title":"A Survey on Mamba Architecture for Vision Applications","date":"2025-02-11","arxiv_id":"2502.07161","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-video-understanding-deep-neural","title":"Enhancing Video Understanding: Deep Neural Networks for Spatiotemporal Analysis","date":"2025-02-11","arxiv_id":"2502.07277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-video-analytics-in-cloud-edge","title":"A Survey on Video Analytics in Cloud-Edge-Terminal Collaborative Systems","date":"2025-02-10","arxiv_id":"2502.06581","repositories_listed":0,"syntology":null},{"url":null,"slug":"cos-chain-of-shot-prompting-for-long-video","title":"CoS: Chain-of-Shot Prompting for Long Video Understanding","date":"2025-02-10","arxiv_id":"2502.06428","repositories_listed":0,"syntology":null},{"url":null,"slug":"worldsense-evaluating-real-world-omnimodal","title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","date":"2025-02-06","arxiv_id":"2502.04326","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-decade-of-action-quality-assessment-largest","title":"A Decade of Action Quality Assessment: Largest Systematic Survey of Trends, Challenges, and Future Directions","date":"2025-02-05","arxiv_id":"2502.02817","repositories_listed":0,"syntology":null},{"url":null,"slug":"maxinfo-a-training-free-key-frame-selection","title":"MaxInfo: A Training-Free Key-Frame Selection Method Using Maximum Volume for Enhanced Video Understanding","date":"2025-02-05","arxiv_id":"2502.03183","repositories_listed":0,"syntology":null},{"url":null,"slug":"lv-xattn-distributed-cross-attention-for-long","title":"LV-XAttn: Distributed Cross-Attention for Long Visual Inputs in Multimodal Large Language Models","date":"2025-02-04","arxiv_id":"2502.02406","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-role-of-explicit-temporal","title":"Exploring the Role of Explicit Temporal Modeling in Multimodal Large Language Models for Video Understanding","date":"2025-01-28","arxiv_id":"2501.16786","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-long-videos-via-llm-powered","title":"Understanding Long Videos via LLM-Powered Entity Relation Graphs","date":"2025-01-27","arxiv_id":"2501.15953","repositories_listed":0,"syntology":null},{"url":null,"slug":"humanomni-a-large-vision-speech-language","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","date":"2025-01-25","arxiv_id":"2501.15111","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-preference-optimization-for-long","title":"Temporal Preference Optimization for Long-Form Video Understanding","date":"2025-01-23","arxiv_id":"2501.13919","repositories_listed":0,"syntology":null},{"url":null,"slug":"hfgcn-hypergraph-fusion-graph-convolutional","title":"HFGCN:Hypergraph Fusion Graph Convolutional Networks for Skeleton-Based Action Recognition","date":"2025-01-19","arxiv_id":"2501.11007","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-rgpt-unifying-image-and-video-region","title":"Omni-RGPT: Unifying Image and Video Region-level Understanding via Token Marks","date":"2025-01-14","arxiv_id":"2501.08326","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-quality-assessment-for-online","title":"Video Quality Assessment for Online Processing: From Spatial to Temporal Sampling","date":"2025-01-13","arxiv_id":"2501.07087","repositories_listed":0,"syntology":null},{"url":null,"slug":"x-lebench-a-benchmark-for-extremely-long","title":"X-LeBench: A Benchmark for Extremely Long Egocentric Video Understanding","date":"2025-01-12","arxiv_id":"2501.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-shark-tracking-and-biometrics-from","title":"Zero-shot Shark Tracking and Biometrics from Aerial Imagery","date":"2025-01-10","arxiv_id":"2501.05717","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-octopus-unlocking-instruction-driven","title":"LLaVA-Octopus: Unlocking Instruction-Driven Adaptive Projector Fusion for Video Understanding","date":"2025-01-09","arxiv_id":"2501.05067","repositories_listed":0,"syntology":null},{"url":null,"slug":"longvitu-instruction-tuning-for-long-form","title":"LongViTU: Instruction Tuning for Long-Form Video Understanding","date":"2025-01-09","arxiv_id":"2501.05037","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-mind-palace-structuring","title":"Building a Mind Palace: Structuring Environment-Grounded Semantic Graphs for Effective Long Video Analysis with LLMs","date":"2025-01-08","arxiv_id":"2501.04336","repositories_listed":0,"syntology":null},{"url":null,"slug":"h-mba-hierarchical-mamba-adaptation-for-multi","title":"H-MBA: Hierarchical MamBa Adaptation for Multi-Modal Video Understanding in Autonomous Driving","date":"2025-01-08","arxiv_id":"2501.04302","repositories_listed":0,"syntology":null},{"url":null,"slug":"motionbench-benchmarking-and-improving-fine","title":"MotionBench: Benchmarking and Improving Fine-grained Video Motion Understanding for Vision Language Models","date":"2025-01-06","arxiv_id":"2501.02955","repositories_listed":0,"syntology":null}],"record_sha256":"0ebd8d7712bc39018bfdd353779b75af4141658e38638ffcc46927ff1aa6d3a9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}