{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/8","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":12,"rows_per_page":100,"rows":[701,800],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding/papers/7","next":"/task/video-understanding/papers/9","papers":[{"url":null,"slug":"adacm-2-on-understanding-extremely-long-term-1","title":"AdaCM^2: On Understanding Extremely Long-Term Video with Adaptive Cross-Modality Memory Reduction","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-pre-trained-3d-models-for-point","title":"Adapting Pre-trained 3D Models for Point Cloud Video Understanding via Cross-frame Spatio-temporal Perception","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-motion-aware-video-mllm","title":"Efficient Motion-Aware Video MLLM","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-frame-selection-for-efficient-video","title":"Flexible Frame Selection for Efficient Video Reasoning","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarq-task-aware-hierarchical-q-former-for","title":"HierarQ: Task-Aware Hierarchical Q-Former for Enhanced Video Understanding","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"humocon-concept-discovery-for-human-motion","title":"HuMoCon: Concept Discovery for Human Motion Understanding","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"q-bench-video-benchmark-the-video-quality","title":"Q-Bench-Video: Benchmark the Video Quality Understanding of LMMs","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"veu-bench-towards-comprehensive-understanding","title":"VEU-Bench: Towards Comprehensive Understanding of Video Editing","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-language-model-pretraining-with-spatio","title":"Video Language Model Pretraining with Spatio-temporal Masking","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-temporal-action-9","title":"Weakly Supervised Temporal Action Localization via Dual-Prior Collaborative Learning Guided by Multimodal Large Language Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-videoagent-persistent-memory-from","title":"Embodied VideoAgent: Persistent Memory from Egocentric Videos and Embodied Sensors Enables Dynamic Scene Understanding","date":"2024-12-31","arxiv_id":"2501.00358","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-video-text-retrieval-a-new","title":"CaReBench: A Fine-Grained Benchmark for Video Captioning and Retrieval","date":"2024-12-31","arxiv_id":"2501.00513","repositories_listed":0,"syntology":null},{"url":null,"slug":"ov-hhir-open-vocabulary-human-interaction","title":"OV-HHIR: Open Vocabulary Human Interaction Recognition Using Cross-modal Integration of Large Language Models","date":"2024-12-31","arxiv_id":"2501.00432","repositories_listed":0,"syntology":null},{"url":null,"slug":"mvtamperbench-evaluating-robustness-of-vision","title":"MVTamperBench: Evaluating Robustness of Vision-Language Models","date":"2024-12-27","arxiv_id":"2412.19794","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceive-query-reason-enhancing-video-qa-with","title":"Perceive, Query & Reason: Enhancing Video QA with Question-Guided Temporal Queries","date":"2024-12-26","arxiv_id":"2412.19304","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-domain-incremental-learning-for-human","title":"Video Domain Incremental Learning for Human Action Recognition in Home Environments","date":"2024-12-22","arxiv_id":"2412.16946","repositories_listed":0,"syntology":null},{"url":null,"slug":"focuschat-text-guided-long-video","title":"FocusChat: Text-guided Long Video Understanding via Spatiotemporal Information Filtering","date":"2024-12-17","arxiv_id":"2412.12833","repositories_listed":0,"syntology":null},{"url":null,"slug":"shotvl-human-centric-highlight-frame","title":"ShotVL: Human-Centric Highlight Frame Retrieval via Language Queries","date":"2024-12-17","arxiv_id":"2412.12675","repositories_listed":0,"syntology":null},{"url":null,"slug":"cg-bench-clue-grounded-question-answering","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","date":"2024-12-16","arxiv_id":"2412.12075","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-trec-2024-medical-video-question","title":"Overview of TREC 2024 Medical Video Question Answering (MedVidQA) Track","date":"2024-12-15","arxiv_id":"2412.11056","repositories_listed":0,"syntology":null},{"url":null,"slug":"apollo-an-exploration-of-video-understanding","title":"Apollo: An Exploration of Video Understanding in Large Multimodal Models","date":"2024-12-13","arxiv_id":"2412.10360","repositories_listed":0,"syntology":null},{"url":null,"slug":"iqvic-in-context-question-adaptive-vision","title":"IQViC: In-context, Question Adaptive Vision Compressor for Long-term Video Understanding LMMs","date":"2024-12-13","arxiv_id":"2412.09907","repositories_listed":0,"syntology":null},{"url":null,"slug":"pvc-progressive-visual-token-compression-for","title":"PVC: Progressive Visual Token Compression for Unified Image and Video Processing in Large Vision-Language Models","date":"2024-12-12","arxiv_id":"2412.09613","repositories_listed":0,"syntology":null},{"url":null,"slug":"vca-video-curious-agent-for-long-video","title":"VCA: Video Curious Agent for Long Video Understanding","date":"2024-12-12","arxiv_id":"2412.10471","repositories_listed":0,"syntology":null},{"url":null,"slug":"vicas-a-dataset-for-combining-holistic-and","title":"ViCaS: A Dataset for Combining Holistic and Pixel-level Video Understanding using Captions with Grounded Segmentation","date":"2024-12-12","arxiv_id":"2412.09754","repositories_listed":0,"syntology":null},{"url":null,"slug":"coef-vq-cost-efficient-video-quality","title":"COEF-VQ: Cost-Efficient Video Quality Understanding through a Cascaded Multimodal LLM Framework","date":"2024-12-11","arxiv_id":"2412.10435","repositories_listed":0,"syntology":null},{"url":null,"slug":"3dsrbench-a-comprehensive-3d-spatial","title":"3DSRBench: A Comprehensive 3D Spatial Reasoning Benchmark","date":"2024-12-10","arxiv_id":"2412.07825","repositories_listed":0,"syntology":null},{"url":null,"slug":"gexia-granularity-expansion-and-iterative","title":"GEXIA: Granularity Expansion and Iterative Approximation for Scalable Multi-grained Video-language Learning","date":"2024-12-10","arxiv_id":"2412.07704","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-contrastive-learning-for-video","title":"Multi-Scale Contrastive Learning for Video Temporal Grounding","date":"2024-12-10","arxiv_id":"2412.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-long-video-understanding-via-fine","title":"Towards Long Video Understanding via Fine-detailed Video Story Generation","date":"2024-12-09","arxiv_id":"2412.06182","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-boxes-mask-guided-spatio-temporal","title":"Beyond Boxes: Mask-Guided Spatio-Temporal Feature Aggregation for Video Object Detection","date":"2024-12-06","arxiv_id":"2412.04915","repositories_listed":0,"syntology":null},{"url":null,"slug":"espresso-high-compression-for-rich-extraction","title":"Espresso: High Compression For Rich Extraction From Videos for Your Vision-Language Model","date":"2024-12-06","arxiv_id":"2412.04729","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidhalluc-evaluating-temporal-hallucinations","title":"VidHalluc: Evaluating Temporal Hallucinations in Multimodal Large Language Models for Video Understanding","date":"2024-12-04","arxiv_id":"2412.03735","repositories_listed":0,"syntology":null},{"url":null,"slug":"progress-aware-video-frame-captioning","title":"Progress-Aware Video Frame Captioning","date":"2024-12-03","arxiv_id":"2412.02071","repositories_listed":0,"syntology":null},{"url":null,"slug":"seal-semantic-attention-learning-for-long","title":"SEAL: Semantic Attention Learning for Long Video Representation","date":"2024-12-02","arxiv_id":"2412.01798","repositories_listed":0,"syntology":null},{"url":null,"slug":"videosavi-self-aligned-video-language-models","title":"VideoSAVi: Self-Aligned Video Language Models without Human Supervision","date":"2024-12-01","arxiv_id":"2412.00624","repositories_listed":0,"syntology":null},{"url":null,"slug":"vista-enhancing-long-duration-and-high","title":"VISTA: Enhancing Long-Duration and High-Resolution Video Understanding by Video Spatiotemporal Augmentation","date":"2024-12-01","arxiv_id":"2412.00927","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-every-frame-all-at-once-video-ma-2-mba","title":"Look Every Frame All at Once: Video-Ma$^2$mba for Efficient Long-form Video Understanding with Multi-Axis Gradient Checkpointing","date":"2024-11-29","arxiv_id":"2411.19460","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-test-2024-challenge-summary-and-a","title":"Perception Test 2024: Challenge Summary and a Novel Hour-Long VideoQA Benchmark","date":"2024-11-29","arxiv_id":"2411.19941","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-enhancing-video-llms-compositional","title":"STEP: Enhancing Video-LLMs' Compositional Reasoning by Spatio-Temporal Graph-guided Self-Training","date":"2024-11-29","arxiv_id":"2412.00161","repositories_listed":0,"syntology":null},{"url":null,"slug":"saven-vid-synergistic-audio-visual","title":"SAVEn-Vid: Synergistic Audio-Visual Integration for Enhanced Understanding in Long Video Context","date":"2024-11-25","arxiv_id":"2411.16213","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewind-understanding-long-videos-with","title":"ReWind: Understanding Long Videos with Instructed Learnable Memory","date":"2024-11-23","arxiv_id":"2411.15556","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-training-dynamic-token-merging-for","title":"Beyond Training: Dynamic Token Merging for Zero-Shot Video Understanding","date":"2024-11-21","arxiv_id":"2411.14401","repositories_listed":0,"syntology":null},{"url":"/paper/extending-video-masked-autoencoders-to-128-1","slug":"extending-video-masked-autoencoders-to-128-1","title":"Extending Video Masked Autoencoders to 128 frames","date":"2024-11-20","arxiv_id":"2411.13683","repositories_listed":0,"syntology":null},{"url":null,"slug":"principles-of-visual-tokens-for-efficient","title":"Principles of Visual Tokens for Efficient Video Understanding","date":"2024-11-20","arxiv_id":"2411.13626","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoautoarena-an-automated-arena-for","title":"VideoAutoArena: An Automated Arena for Evaluating Large Multimodal Models in Video Analysis through User Simulation","date":"2024-11-20","arxiv_id":"2411.13281","repositories_listed":0,"syntology":null},{"url":null,"slug":"adacm-2-on-understanding-extremely-long-term","title":"AdaCM$^2$: On Understanding Extremely Long-Term Video with Adaptive Cross-Modality Memory Reduction","date":"2024-11-19","arxiv_id":"2411.12593","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynfocus-dynamic-cooperative-network-empowers","title":"DynFocus: Dynamic Cooperative Network Empowers LLMs with Video Understanding","date":"2024-11-19","arxiv_id":"2411.12355","repositories_listed":0,"syntology":null},{"url":null,"slug":"vibe-a-text-to-video-benchmark-for-evaluating","title":"ViBe: A Text-to-Video Benchmark for Evaluating Hallucination in Large Multimodal Models","date":"2024-11-16","arxiv_id":"2411.10867","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-mllms-guide-weakly-supervised-temporal","title":"Can MLLMs Guide Weakly-Supervised Temporal Action Localization Tasks?","date":"2024-11-13","arxiv_id":"2411.08466","repositories_listed":0,"syntology":null},{"url":null,"slug":"evqascore-efficient-video-question-answering","title":"EVQAScore: Efficient Video Question Answering Data Evaluation","date":"2024-11-11","arxiv_id":"2411.06908","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-rwkv-video-action-recognition-based","title":"Video RWKV:Video Action Recognition Based RWKV","date":"2024-11-08","arxiv_id":"2411.05636","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-video-summarization-by","title":"Personalized Video Summarization by Multimodal Video Understanding","date":"2024-11-05","arxiv_id":"2411.03531","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-token-merging-for-long-form-video","title":"Video Token Merging for Long-form Video Understanding","date":"2024-10-31","arxiv_id":"2410.23782","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-action-recognition-in-surveillance","title":"Zero-Shot Action Recognition in Surveillance Videos","date":"2024-10-28","arxiv_id":"2410.21113","repositories_listed":0,"syntology":null},{"url":null,"slug":"exocentric-to-egocentric-transfer-for-action","title":"Egocentric and Exocentric Methods: A Short Survey","date":"2024-10-27","arxiv_id":"2410.20621","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-video-understanding-agent-enhancing","title":"Adaptive Video Understanding Agent: Enhancing efficiency with dynamic frame sampling and feedback-driven reasoning","date":"2024-10-26","arxiv_id":"2410.20252","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-lvlms-describe-videos-like-humans-a-five","title":"FIOVA: A Multi-Annotator Benchmark for Human-Aligned Video Captioning","date":"2024-10-20","arxiv_id":"2410.15270","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextdet-temporal-action-detection-with","title":"ContextDet: Temporal Action Detection with Adaptive Context Aggregation","date":"2024-10-20","arxiv_id":"2410.15279","repositories_listed":0,"syntology":null},{"url":null,"slug":"eva-an-embodied-world-model-for-future-video","title":"EVA: An Embodied World Model for Future Video Anticipation","date":"2024-10-20","arxiv_id":"2410.15461","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-every-frame-matter-continuous-video","title":"Making Every Frame Matter: Continuous Video Understanding for Large Models via Adaptive State Modeling","date":"2024-10-19","arxiv_id":"2410.14993","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-action-localization-via-the","title":"Zero-shot Action Localization via the Confidence of Large Vision-Language Models","date":"2024-10-18","arxiv_id":"2410.14340","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidcompress-memory-enhanced-temporal","title":"VidCompress: Memory-Enhanced Temporal Compression for Video Understanding in Large Language Models","date":"2024-10-15","arxiv_id":"2410.11417","repositories_listed":0,"syntology":null},{"url":null,"slug":"vifi-reid-a-two-stream-vision-wifi-multimodal","title":"ViFi-ReID: A Two-Stream Vision-WiFi Multimodal Approach for Person Re-identification","date":"2024-10-13","arxiv_id":"2410.09875","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-video-language-foundation-models","title":"Prompting Video-Language Foundation Models with Domain-specific Fine-grained Heuristics for Video Question Answering","date":"2024-10-12","arxiv_id":"2410.09380","repositories_listed":0,"syntology":null},{"url":"/paper/tvbench-redesigning-video-language-evaluation","slug":"tvbench-redesigning-video-language-evaluation","title":"TVBench: Redesigning Video-Language Evaluation","date":"2024-10-10","arxiv_id":"2410.07752","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multimodal-llm-for-detailed-and","title":"Enhancing Multimodal LLM for Detailed and Accurate Video Captioning using Multi-Round Preference Optimization","date":"2024-10-09","arxiv_id":"2410.06682","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-ego-towards-building-egocentric-multimodal","title":"MM-Ego: Towards Building Egocentric Multimodal LLMs","date":"2024-10-09","arxiv_id":"2410.07177","repositories_listed":0,"syntology":null},{"url":null,"slug":"auroracap-efficient-performant-video-detailed","title":"AuroraCap: Efficient, Performant Video Detailed Captioning and a New Benchmark","date":"2024-10-04","arxiv_id":"2410.03051","repositories_listed":0,"syntology":null},{"url":null,"slug":"frame-voyager-learning-to-query-frames-for","title":"Frame-Voyager: Learning to Query Frames for Video Large Language Models","date":"2024-10-04","arxiv_id":"2410.03226","repositories_listed":0,"syntology":null},{"url":"/paper/airletters-an-open-video-dataset-of","slug":"airletters-an-open-video-dataset-of","title":"AirLetters: An Open Video Dataset of Characters Drawn in the Air","date":"2024-10-03","arxiv_id":"2410.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"dtvlt-a-multi-modal-diverse-text-benchmark","title":"DTVLT: A Multi-modal Diverse Text Benchmark for Visual Language Tracking Based on LLM","date":"2024-10-03","arxiv_id":"2410.02492","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-action-spotting-in","title":"Deep learning for action spotting in association football videos","date":"2024-10-02","arxiv_id":"2410.01304","repositories_listed":0,"syntology":null},{"url":"/paper/mm1-5-methods-analysis-insights-from","slug":"mm1-5-methods-analysis-insights-from","title":"MM1.5: Methods, Analysis & Insights from Multimodal LLM Fine-tuning","date":"2024-09-30","arxiv_id":"2409.20566","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-bench-video-benchmarking-the-video-quality","title":"Q-Bench-Video: Benchmarking the Video Quality Understanding of LMMs","date":"2024-09-30","arxiv_id":"2409.20063","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-context-window-extension-a-new","title":"Visual Context Window Extension: A New Perspective for Long Video Understanding","date":"2024-09-30","arxiv_id":"2409.20018","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal2seq-a-unified-framework-for-temporal","title":"Temporal2Seq: A Unified Framework for Temporal Video Understanding Tasks","date":"2024-09-27","arxiv_id":"2409.18478","repositories_listed":0,"syntology":null},{"url":null,"slug":"eagle-egocentric-aggregated-language-video","title":"EAGLE: Egocentric AGgregated Language-video Engine","date":"2024-09-26","arxiv_id":"2409.17523","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm4brain-training-a-large-language-model-for","title":"LLM4Brain: Training a Large Language Model for Brain Video Understanding","date":"2024-09-26","arxiv_id":"2409.17987","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-clip-count-stars-an-empirical-study-on","title":"Can CLIP Count Stars? An Empirical Study on Quantity Bias in CLIP","date":"2024-09-23","arxiv_id":"2409.15035","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-place-solution-to-the-multiple-choice","title":"First Place Solution to the Multiple-choice Video QA Track of The Second Perception Test Challenge","date":"2024-09-20","arxiv_id":"2409.13538","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-child-inclusive-clinical-video","title":"Towards Child-Inclusive Clinical Video Understanding for Autism Spectrum Disorder","date":"2024-09-20","arxiv_id":"2409.13606","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-action-recognition-on-hard-to","title":"Interpretable Action Recognition on Hard to Classify Actions","date":"2024-09-19","arxiv_id":"2409.13091","repositories_listed":0,"syntology":null},{"url":null,"slug":"amego-active-memory-from-long-egocentric","title":"AMEGO: Active Memory from long EGOcentric videos","date":"2024-09-17","arxiv_id":"2409.10917","repositories_listed":0,"syntology":null},{"url":null,"slug":"havana-hierarchical-stochastic-neighbor","title":"HAVANA: Hierarchical stochastic neighbor embedding for Accelerated Video ANnotAtions","date":"2024-09-16","arxiv_id":"2409.10641","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-long-video-understanding-via","title":"Enhancing Long Video Understanding via Hierarchical Event-Based Memory","date":"2024-09-10","arxiv_id":"2409.06299","repositories_listed":0,"syntology":null},{"url":null,"slug":"vidlpro-a-underline-vid-eo-underline-l","title":"VidLPRO: A $\\underline{Vid}$eo-$\\underline{L}$anguage $\\underline{P}$re-training Framework for $\\underline{Ro}$botic and Laparoscopic Surgery","date":"2024-09-07","arxiv_id":"2409.04732","repositories_listed":0,"syntology":null},{"url":null,"slug":"tc-llava-rethinking-the-transfer-from-image","title":"TC-LLaVA: Rethinking the Transfer from Image to Video Understanding with Temporal Considerations","date":"2024-09-05","arxiv_id":"2409.03206","repositories_listed":0,"syntology":null},{"url":null,"slug":"videollamb-long-context-video-understanding","title":"VideoLLaMB: Long-context Video Understanding with Recurrent Memory Bridges","date":"2024-09-02","arxiv_id":"2409.01071","repositories_listed":0,"syntology":null},{"url":null,"slug":"stimuvar-spatiotemporal-stimuli-aware-video","title":"StimuVAR: Spatiotemporal Stimuli-aware Video Affective Reasoning with Multimodal Large Language Models","date":"2024-08-31","arxiv_id":"2409.00304","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamlining-forest-wildfire-surveillance-ai","title":"Streamlining Forest Wildfire Surveillance: AI-Enhanced UAVs Utilizing the FLAME Aerial Video Dataset for Lightweight and Efficient Monitoring","date":"2024-08-31","arxiv_id":"2409.00510","repositories_listed":0,"syntology":null},{"url":null,"slug":"dlm-vmtl-a-double-layer-mapper-for","title":"DLM-VMTL:A Double Layer Mapper for heterogeneous data video Multi-task prompt learning","date":"2024-08-29","arxiv_id":"2408.16195","repositories_listed":0,"syntology":null},{"url":null,"slug":"kangaroo-a-powerful-video-language-model","title":"Kangaroo: A Powerful Video-Language Model Supporting Long-context Video Input","date":"2024-08-28","arxiv_id":"2408.15542","repositories_listed":0,"syntology":null},{"url":null,"slug":"attend-fusion-efficient-audio-visual-fusion","title":"Attend-Fusion: Efficient Audio-Visual Fusion for Video Classification","date":"2024-08-26","arxiv_id":"2408.14441","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmm-vqa-advancing-video-quality-assessment","title":"LMM-VQA: Advancing Video Quality Assessment with Large Multimodal Models","date":"2024-08-26","arxiv_id":"2408.14008","repositories_listed":0,"syntology":null},{"url":null,"slug":"flatten-video-action-recognition-is-an-image","title":"Flatten: Video Action Recognition is an Image Classification task","date":"2024-08-17","arxiv_id":"2408.09220","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangle-and-denoise-tackling-context","title":"Disentangle and denoise: Tackling context misalignment for video moment retrieval","date":"2024-08-14","arxiv_id":"2408.07600","repositories_listed":0,"syntology":null},{"url":null,"slug":"spherical-world-locking-for-audio-visual","title":"Spherical World-Locking for Audio-Visual Localization in Egocentric Videos","date":"2024-08-09","arxiv_id":"2408.05364","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02421","title":"FE-Adapter: Adapting Image-based Emotion Classifiers to Videos","date":"2024-08-05","arxiv_id":"2408.02421","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00365","title":"Multimodal Fusion and Coherence Modeling for Video Topic Segmentation","date":"2024-08-01","arxiv_id":"2408.00365","repositories_listed":0,"syntology":null}],"record_sha256":"f29a9bd9204c3d51022b5dd7988ef0aa4d7dcc1bbc45375bc92be1435a2be71a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}