{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-understanding/papers/9","list_of":"/task/video-understanding","task":"Video Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":12,"rows_per_page":100,"rows":[801,900],"of":1149,"counts":{"archive_papers_tagged":1149,"with_a_code_link":542,"where_syntology_ran_a_sample":218,"not_listed_spam_title":0,"listed":1149,"listed_where_code_ran":218,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-understanding","prev":"/task/video-understanding/papers/8","next":"/task/video-understanding/papers/10","papers":[{"url":null,"slug":"adversarial-robustness-in-rgb-skeleton-action","title":"Adversarial Robustness in RGB-Skeleton Action Recognition: Leveraging Attention Modality Reweighter","date":"2024-07-29","arxiv_id":"2407.19981","repositories_listed":0,"syntology":null},{"url":null,"slug":"ego-vpa-egocentric-video-understanding-with","title":"Ego-VPA: Egocentric Video Understanding with Parameter-efficient Adaptation","date":"2024-07-28","arxiv_id":"2407.19520","repositories_listed":0,"syntology":null},{"url":null,"slug":"wolf-captioning-everything-with-a-world","title":"Wolf: Captioning Everything with a World Summarization Framework","date":"2024-07-26","arxiv_id":"2407.18908","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-training-for-improved-grounding","title":"Audio-visual training for improved grounding in video-text LLMs","date":"2024-07-21","arxiv_id":"2407.15046","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-video-text-understanding-retrieval","title":"Rethinking Video-Text Understanding: Retrieval from Counterfactually Augmented Data","date":"2024-07-18","arxiv_id":"2407.13094","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-multi-label-video","title":"Open Vocabulary Multi-Label Video Classification","date":"2024-07-12","arxiv_id":"2407.09073","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-image-to-video-adaptation-an","title":"Rethinking Image-to-Video Adaptation: An Object-centric Perspective","date":"2024-07-09","arxiv_id":"2407.06871","repositories_listed":0,"syntology":null},{"url":null,"slug":"omchat-a-recipe-to-train-multimodal-language","title":"OmChat: A Recipe to Train Multimodal Language Models with Strong Long Context and Video Understanding","date":"2024-07-06","arxiv_id":"2407.04923","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyvideollm-towards-large-scale-video","title":"KeyVideoLLM: Towards Large-scale Video Keyframe Selection","date":"2024-07-03","arxiv_id":"2407.03104","repositories_listed":0,"syntology":null},{"url":null,"slug":"https-arxiv-org-abs-2407-00634","title":"https://arxiv.org/abs/2407.00634","date":"2024-07-02","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-watermarking-safeguarding-your-video","title":"Video Watermarking: Safeguarding Your Video from (Unauthorized) Annotations by Video-based LLMs","date":"2024-07-02","arxiv_id":"2407.02411","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-long-form-video-understanding","title":"Zero-Shot Long-Form Video Understanding through Screenplay","date":"2024-06-25","arxiv_id":"2406.17309","repositories_listed":0,"syntology":null},{"url":null,"slug":"videohallucer-evaluating-intrinsic-and","title":"VideoHallucer: Evaluating Intrinsic and Extrinsic Hallucinations in Large Video-Language Models","date":"2024-06-24","arxiv_id":"2406.16338","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-salmonn-speech-enhanced-audio-visual","title":"video-SALMONN: Speech-Enhanced Audio-Visual Large Language Models","date":"2024-06-22","arxiv_id":"2406.15704","repositories_listed":0,"syntology":null},{"url":null,"slug":"gvt2rpm-an-empirical-study-for-general-video","title":"GVT2RPM: An Empirical Study for General Video Transformer Adaptation to Remote Physiological Measurement","date":"2024-06-19","arxiv_id":"2406.13136","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-holistic-language-video","title":"Towards Holistic Language-video Representation: the language model-enhanced MSR-Video to Text Dataset","date":"2024-06-19","arxiv_id":"2406.13809","repositories_listed":0,"syntology":null},{"url":null,"slug":"drvideo-document-retrieval-based-long-video","title":"DrVideo: Document Retrieval Based Long Video Understanding","date":"2024-06-18","arxiv_id":"2406.12846","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-mitigation-prompts-long-term","title":"Hallucination Mitigation Prompts Long-term Video Understanding","date":"2024-06-17","arxiv_id":"2406.11333","repositories_listed":0,"syntology":null},{"url":null,"slug":"velociti-can-video-language-models-bind","title":"VELOCITI: Benchmarking Video-Language Compositional Reasoning with Strict Entailment","date":"2024-06-16","arxiv_id":"2406.10889","repositories_listed":0,"syntology":null},{"url":"/paper/gpt-4o-visual-perception-performance-of","slug":"gpt-4o-visual-perception-performance-of","title":"GPT-4o: Visual perception performance of multimodal large language models in piglet activity understanding","date":"2024-06-14","arxiv_id":"2406.09781","repositories_listed":0,"syntology":null},{"url":null,"slug":"localizing-events-in-videos-with-multimodal","title":"Localizing Events in Videos with Multimodal Queries","date":"2024-06-14","arxiv_id":"2406.10079","repositories_listed":0,"syntology":null},{"url":null,"slug":"llavidal-benchmarking-large-language-vision","title":"LLAVIDAL: A Large LAnguage VIsion Model for Daily Activities of Living","date":"2024-06-13","arxiv_id":"2406.09390","repositories_listed":0,"syntology":null},{"url":null,"slug":"fewer-tokens-and-fewer-videos-extending-video","title":"Fewer Tokens and Fewer Videos: Extending Video Understanding Abilities in Large Vision-Language Models","date":"2024-06-12","arxiv_id":"2406.08024","repositories_listed":0,"syntology":null},{"url":null,"slug":"memsvd-long-range-temporal-structure","title":"MeMSVD: Long-Range Temporal Structure Capturing Using Incremental SVD","date":"2024-06-11","arxiv_id":"2406.07191","repositories_listed":0,"syntology":null},{"url":null,"slug":"1st-place-winner-of-the-2024-pixel-level","title":"1st Place Winner of the 2024 Pixel-level Video Understanding in the Wild (CVPR'24 PVUW) Challenge in Video Panoptic Segmentation and Best Long Video Consistency of Video Semantic Segmentation","date":"2024-06-08","arxiv_id":"2406.05352","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-segmentation-on-vspw-dataset-through-2","title":"Semantic Segmentation on VSPW Dataset through Masked Video Consistency","date":"2024-06-07","arxiv_id":"2406.04979","repositories_listed":0,"syntology":null},{"url":null,"slug":"3rd-place-solution-for-pvuw-challenge-2024","title":"3rd Place Solution for PVUW Challenge 2024: Video Panoptic Segmentation","date":"2024-06-06","arxiv_id":"2406.04002","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-language-video-time-pre-training","title":"Contrastive Language Video Time Pre-training","date":"2024-06-04","arxiv_id":"2406.02631","repositories_listed":0,"syntology":null},{"url":null,"slug":"2nd-place-solution-for-pvuw-challenge-2024","title":"2nd Place Solution for PVUW Challenge 2024: Video Panoptic Segmentation","date":"2024-06-01","arxiv_id":"2406.00500","repositories_listed":0,"syntology":null},{"url":null,"slug":"henasy-learning-to-assemble-scene-entities","title":"HENASY: Learning to Assemble Scene-Entities for Egocentric Video-Language Model","date":"2024-06-01","arxiv_id":"2406.00307","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-grounding-of-activities-using","title":"Temporal Grounding of Activities using Multimodal Large Language Models","date":"2024-05-30","arxiv_id":"2407.06157","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-action-recognition-a-contrastive","title":"Hierarchical Action Recognition: A Contrastive Video-Language Approach with Hierarchical Interactions","date":"2024-05-28","arxiv_id":"2405.17729","repositories_listed":0,"syntology":null},{"url":"/paper/mmctagent-multi-modal-critical-thinking-agent","slug":"mmctagent-multi-modal-critical-thinking-agent","title":"MMCTAgent: Multi-modal Critical Thinking Agent Framework for Complex Visual Reasoning","date":"2024-05-28","arxiv_id":"2405.18358","repositories_listed":0,"syntology":null},{"url":"/paper/streaming-long-video-understanding-with-large","slug":"streaming-long-video-understanding-with-large","title":"Streaming Long Video Understanding with Large Language Models","date":"2024-05-25","arxiv_id":"2405.16009","repositories_listed":0,"syntology":null},{"url":null,"slug":"mamba4d-efficient-long-sequence-point-cloud","title":"MAMBA4D: Efficient Long-Sequence Point Cloud Video Understanding with Disentangled Spatial-Temporal State Space Models","date":"2024-05-23","arxiv_id":"2405.14338","repositories_listed":0,"syntology":null},{"url":null,"slug":"anticipating-object-state-changes","title":"Anticipating Object State Changes in Long Procedural Videos","date":"2024-05-21","arxiv_id":"2405.12789","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-spatio-temporal-action","title":"Open-Vocabulary Spatio-Temporal Action Detection","date":"2024-05-17","arxiv_id":"2405.10832","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-in-deploying-long-context","title":"Challenges in Deploying Long-Context Transformers: A Theoretical Peak Performance Analysis","date":"2024-05-14","arxiv_id":"2405.08944","repositories_listed":0,"syntology":null},{"url":"/paper/cinepile-a-long-video-question-answering","slug":"cinepile-a-long-video-question-answering","title":"CinePile: A Long Video Question Answering Dataset and Benchmark","date":"2024-05-14","arxiv_id":"2405.08813","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-motion-understanding-in-large-scale","title":"Global Motion Understanding in Large-Scale Video Object Segmentation","date":"2024-05-11","arxiv_id":"2405.07031","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-enhanced-zero-shot-video-captioning","title":"RETTA: Retrieval-Enhanced Test-Time Adaptation for Zero-Shot Video Captioning","date":"2024-05-11","arxiv_id":"2405.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-backbones-for-deep-video-action","title":"A Survey on Backbones for Deep Video Action Recognition","date":"2024-05-09","arxiv_id":"2405.05584","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-of-thought-step-by-step-video-reasoning","title":"Video-of-Thought: Step-by-Step Video Reasoning from Perception to Cognition","date":"2024-05-07","arxiv_id":"2501.03230","repositories_listed":0,"syntology":null},{"url":null,"slug":"complex-video-reasoning-and-robustness","title":"How Good is my Video LMM? Complex Video Reasoning and Robustness Evaluation Suite for Video-LMMs","date":"2024-05-06","arxiv_id":"2405.03690","repositories_listed":0,"syntology":null},{"url":null,"slug":"worldqa-multimodal-world-knowledge-in-videos","title":"WorldQA: Multimodal World Knowledge in Videos through Long-Chain Reasoning","date":"2024-05-06","arxiv_id":"2405.03272","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-text-to-video-retrieval-from-image","title":"Learning text-to-video retrieval from image captioning","date":"2024-04-26","arxiv_id":"2404.17498","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-set-video-based-facial-expression","title":"Open-Set Video-based Facial Expression Recognition with Human Expression-sensitive Prompting","date":"2024-04-26","arxiv_id":"2404.17100","repositories_listed":0,"syntology":null},{"url":null,"slug":"ipad-industrial-process-anomaly-detection","title":"IPAD: Industrial Process Anomaly Detection Dataset","date":"2024-04-23","arxiv_id":"2404.15033","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-image-to-video-what-do-we-need-in","title":"From Image to Video, what do we need in multimodal LLMs?","date":"2024-04-18","arxiv_id":"2404.11865","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-transformer-based-model-for-the-prediction","title":"A Transformer-Based Model for the Prediction of Human Gaze Behavior on Videos","date":"2024-04-10","arxiv_id":"2404.07351","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaze-guided-graph-neural-network-for-action","title":"Gaze-Guided Graph Neural Network for Action Anticipation Conditioned on Intention","date":"2024-04-10","arxiv_id":"2404.07347","repositories_listed":0,"syntology":null},{"url":null,"slug":"koala-key-frame-conditioned-long-video-llm","title":"Koala: Key frame-conditioned long video-LLM","date":"2024-04-05","arxiv_id":"2404.04346","repositories_listed":0,"syntology":null},{"url":null,"slug":"biovl-qr-egocentric-biochemical-video-and","title":"BioVL-QR: Egocentric Biochemical Vision-and-Language Dataset Using Micro QR Codes","date":"2024-04-04","arxiv_id":"2404.03161","repositories_listed":0,"syntology":null},{"url":null,"slug":"ow-viscap-open-world-video-instance","title":"OW-VISCapTor: Abstractors for Open-World Video Instance Segmentation and Captioning","date":"2024-04-04","arxiv_id":"2404.03657","repositories_listed":0,"syntology":null},{"url":null,"slug":"instrument-tissue-interaction-detection","title":"Instrument-tissue Interaction Detection Framework for Surgical Video Understanding","date":"2024-03-30","arxiv_id":"2404.00322","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-human-centric-point","title":"A Unified Framework for Human-centric Point Cloud Video Understanding","date":"2024-03-29","arxiv_id":"2403.20031","repositories_listed":0,"syntology":null},{"url":null,"slug":"avicuna-audio-visual-llm-with-interleaver-and","title":"Empowering LLMs with Pseudo-Untrimmed Videos for Audio-Visual Temporal Understanding","date":"2024-03-24","arxiv_id":"2403.16276","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoagent-a-memory-augmented-multimodal","title":"VideoAgent: A Memory-augmented Multimodal Agent for Video Understanding","date":"2024-03-18","arxiv_id":"2403.11481","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-reimagined-text-to-pose-video-editing","title":"Action Reimagined: Text-to-Pose Video Editing for Dynamic Human Actions","date":"2024-03-11","arxiv_id":"2403.07198","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-backpack-full-of-skills-egocentric-video","title":"A Backpack Full of Skills: Egocentric Video Understanding with Diverse Task Perspectives","date":"2024-03-05","arxiv_id":"2403.03037","repositories_listed":0,"syntology":null},{"url":null,"slug":"moviellm-enhancing-long-video-understanding","title":"MovieLLM: Enhancing Long Video Understanding with AI-Generated Movies","date":"2024-03-03","arxiv_id":"2403.01422","repositories_listed":0,"syntology":null},{"url":null,"slug":"abductive-ego-view-accident-video","title":"Abductive Ego-View Accident Video Understanding for Safe Driving Perception","date":"2024-03-01","arxiv_id":"2403.00436","repositories_listed":0,"syntology":null},{"url":null,"slug":"tv-trees-multimodal-entailment-trees-for","title":"TV-TREES: Multimodal Entailment Trees for Neuro-Symbolic Video Reasoning","date":"2024-02-29","arxiv_id":"2402.19467","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-meet-long-video-advancing-long-video","title":"LLMs Meet Long Video: Advancing Long Video Question Answering with An Interactive Visual Adapter in LLMs","date":"2024-02-21","arxiv_id":"2402.13546","repositories_listed":0,"syntology":null},{"url":null,"slug":"slot-vlm-slowfast-slots-for-video-language","title":"Slot-VLM: SlowFast Slots for Video-Language Modeling","date":"2024-02-20","arxiv_id":"2402.13088","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoprism-a-foundational-visual-encoder-for","title":"VideoPrism: A Foundational Visual Encoder for Video Understanding","date":"2024-02-20","arxiv_id":"2402.13217","repositories_listed":0,"syntology":null},{"url":null,"slug":"system-identification-of-neural-systems-going","title":"Dynamics Based Neural Encoding with Inter-Intra Region Connectivity","date":"2024-02-19","arxiv_id":"2402.12519","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-consolidation-enables-long-context","title":"Memory Consolidation Enables Long-Context Video Understanding","date":"2024-02-08","arxiv_id":"2402.05861","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-generative-ai-and-llm-for-video","title":"A Survey on Generative AI and LLM for Video Generation, Understanding, and Streaming","date":"2024-01-30","arxiv_id":"2404.16038","repositories_listed":0,"syntology":null},{"url":null,"slug":"cutup-and-detect-human-fall-detection-on","title":"Cutup and Detect: Human Fall Detection on Cutup Untrimmed Videos Using a Large Foundational Video Understanding Model","date":"2024-01-29","arxiv_id":"2401.16280","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-missing-modality-in-multimodal","title":"Exploring Missing Modality in Multimodal Egocentric Datasets","date":"2024-01-21","arxiv_id":"2401.11470","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-visually-connect-actions-and","title":"Learning to Visually Connect Actions and their Effects","date":"2024-01-19","arxiv_id":"2401.10805","repositories_listed":0,"syntology":null},{"url":null,"slug":"crossvideo-self-supervised-cross-modal","title":"CrossVideo: Self-supervised Cross-modal Contrastive Learning for Point Cloud Video Understanding","date":"2024-01-17","arxiv_id":"2401.09057","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-2d-temporal-map-diffusion-models","title":"Multi-scale 2D Temporal Map Diffusion Models for Natural Language Video Localization","date":"2024-01-16","arxiv_id":"2401.08232","repositories_listed":0,"syntology":null},{"url":null,"slug":"dr2net-dynamic-reversible-dual-residual","title":"Dr2Net: Dynamic Reversible Dual-Residual Networks for Memory-Efficient Finetuning","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-motion-text-alignment-for-image-to","title":"Enhanced Motion-Text Alignment for Image-to-Video Transfer Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-io-2-scaling-autoregressive-1","title":"Unified-IO 2: Scaling Autoregressive Multimodal Models with Vision Language Audio and Action","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"videogrounding-dino-towards-open-vocabulary","title":"VideoGrounding-DINO: Towards Open-Vocabulary Spatio-Temporal Video Grounding","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-groundingdino-towards-open-vocabulary","title":"Video-GroundingDINO: Towards Open-Vocabulary Spatio-Temporal Video Grounding","date":"2023-12-31","arxiv_id":"2401.00901","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-more-shortcuts-realizing-the-potential-of","title":"No More Shortcuts: Realizing the Potential of Temporal Self-Supervision","date":"2023-12-20","arxiv_id":"2312.13008","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-object-state-changes-in-videos-an","title":"Learning Object State Changes in Videos: An Open-World Perspective","date":"2023-12-19","arxiv_id":"2312.11782","repositories_listed":0,"syntology":null},{"url":"/paper/text-conditioned-resampler-for-long-form","slug":"text-conditioned-resampler-for-long-form","title":"Text-Conditioned Resampler For Long Form Video Understanding","date":"2023-12-19","arxiv_id":"2312.11897","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-optical-hardware","title":"Artificial intelligence optical hardware empowers high-resolution hyperspectral video understanding at 1.2 Tb/s","date":"2023-12-17","arxiv_id":"2312.10639","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-llm-for-video-understanding","title":"Audio-Visual LLM for Video Understanding","date":"2023-12-11","arxiv_id":"2312.06720","repositories_listed":0,"syntology":null},{"url":null,"slug":"movqa-a-benchmark-of-versatile-question","title":"MoVQA: A Benchmark of Versatile Question-Answering for Long-Form Movie Understanding","date":"2023-12-08","arxiv_id":"2312.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-based-video-language-model-for","title":"Retrieval-based Video Language Model for Efficient Long Video Question Answering","date":"2023-12-08","arxiv_id":"2312.04931","repositories_listed":0,"syntology":null},{"url":null,"slug":"hig-hierarchical-interlacement-graph-approach","title":"HIG: Hierarchical Interlacement Graph Approach to Scene Graph Generation in Video Understanding","date":"2023-12-05","arxiv_id":"2312.03050","repositories_listed":0,"syntology":null},{"url":null,"slug":"vaquita-enhancing-alignment-in-llm-assisted","title":"VaQuitA: Enhancing Alignment in LLM-Assisted Video Understanding","date":"2023-12-04","arxiv_id":"2312.02310","repositories_listed":0,"syntology":null},{"url":"/paper/zero-shot-video-question-answering-with","slug":"zero-shot-video-question-answering-with","title":"Zero-Shot Video Question Answering with Procedural Programs","date":"2023-12-01","arxiv_id":"2312.00937","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-video-topic-segmentation-with","title":"Multi-Modal Video Topic Segmentation with Dual-Contrastive Domain Adaptation","date":"2023-11-30","arxiv_id":"2312.00220","repositories_listed":0,"syntology":null},{"url":null,"slug":"spacewalk-18-a-benchmark-for-multimodal-and","title":"Spacewalk-18: A Benchmark for Multimodal and Long-form Procedural Video Understanding","date":"2023-11-30","arxiv_id":"2311.18773","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt4video-a-unified-multimodal-large-language","title":"GPT4Video: A Unified Multimodal Large Language Model for lnstruction-Followed Understanding and Safety-Aware Generation","date":"2023-11-25","arxiv_id":"2311.16511","repositories_listed":0,"syntology":null},{"url":null,"slug":"spot-revisiting-video-language-models-for","title":"SPOT! Revisiting Video-Language Models for Event Understanding","date":"2023-11-21","arxiv_id":"2311.12919","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-still-images-robust-multi-stream","title":"Beyond still images: Temporal features and input variance resilience","date":"2023-11-01","arxiv_id":"2311.00800","repositories_listed":0,"syntology":null},{"url":null,"slug":"probio-a-protocol-guided-multimodal-dataset","title":"ProBio: A Protocol-guided Multimodal Dataset for Molecular Biology Lab","date":"2023-11-01","arxiv_id":"2311.00556","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeetad-adapting-pretrained-vision-language","title":"ZEETAD: Adapting Pretrained Vision-Language Model for Zero-Shot End-to-End Temporal Action Detection","date":"2023-11-01","arxiv_id":"2311.00729","repositories_listed":0,"syntology":null},{"url":"/paper/videoprompter-an-ensemble-of-foundational","slug":"videoprompter-an-ensemble-of-foundational","title":"Videoprompter: an ensemble of foundational models for zero-shot video understanding","date":"2023-10-23","arxiv_id":"2310.15324","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-aware-long-video-localization-and","title":"Query-aware Long Video Localization and Relation Discrimination for Deep Video Understanding","date":"2023-10-19","arxiv_id":"2310.12724","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivegpt4-interpretable-end-to-end-autonomous","title":"DriveGPT4: Interpretable End-to-end Autonomous Driving via Large Language Model","date":"2023-10-02","arxiv_id":"2310.01412","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-3-3d-learning-3d-priors-using-multi-modal","title":"M$^{3}$3D: Learning 3D priors using Multi-Modal Masked Autoencoders for 2D image and video understanding","date":"2023-09-26","arxiv_id":"2309.15313","repositories_listed":0,"syntology":null}],"record_sha256":"16845affeada665fb0441a1e210cb41ae78f880f4e758d525098715d54cf17cc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}