{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-question-answering/papers/4","list_of":"/task/video-question-answering","task":"Video Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":460,"counts":{"archive_papers_tagged":460,"with_a_code_link":250,"where_syntology_ran_a_sample":124,"not_listed_spam_title":0,"listed":460,"listed_where_code_ran":124,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":107,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":107,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-question-answering","prev":"/task/video-question-answering/papers/3","next":"/task/video-question-answering/papers/5","papers":[{"url":null,"slug":"prompting-video-language-foundation-models","title":"Prompting Video-Language Foundation Models with Domain-specific Fine-grained Heuristics for Video Question Answering","date":"2024-10-12","arxiv_id":"2410.09380","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-then-identify-a-general-framework-for","title":"Sample then Identify: A General Framework for Risk Control and Assessment in Multimodal Large Language Models","date":"2024-10-10","arxiv_id":"2410.08174","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multimodal-llm-for-detailed-and","title":"Enhancing Multimodal LLM for Detailed and Accurate Video Captioning using Multi-Round Preference Optimization","date":"2024-10-09","arxiv_id":"2410.06682","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionatlas-a-videoqa-benchmark-for-domain","title":"ActionAtlas: A VideoQA Benchmark for Domain-specialized Action Recognition","date":"2024-10-08","arxiv_id":"2410.05774","repositories_listed":0,"syntology":null},{"url":null,"slug":"frame-voyager-learning-to-query-frames-for","title":"Frame-Voyager: Learning to Query Frames for Video Large Language Models","date":"2024-10-04","arxiv_id":"2410.03226","repositories_listed":0,"syntology":null},{"url":"/paper/video-instruction-tuning-with-synthetic-data","slug":"video-instruction-tuning-with-synthetic-data","title":"Video Instruction Tuning With Synthetic Data","date":"2024-10-03","arxiv_id":"2410.02713","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-dataflywheel-resolving-the-impossible","title":"Video DataFlywheel: Resolving the Impossible Data Trinity in Video-Language Understanding","date":"2024-09-29","arxiv_id":"2409.19532","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-place-solution-to-the-multiple-choice","title":"First Place Solution to the Multiple-choice Video QA Track of The Second Perception Test Challenge","date":"2024-09-20","arxiv_id":"2409.13538","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-guided-self-questioning-and","title":"Uncertainty-Guided Self-Questioning and Answering for Video-Language Alignment","date":"2024-09-17","arxiv_id":"2410.02768","repositories_listed":0,"syntology":null},{"url":null,"slug":"qtg-vqa-question-type-guided-architectural","title":"QTG-VQA: Question-Type-Guided Architectural for VideoQA Systems","date":"2024-09-14","arxiv_id":"2409.09348","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-object-event-graph-representation","title":"Multi-object event graph representation learning for Video Question Answering","date":"2024-09-12","arxiv_id":"2409.07747","repositories_listed":0,"syntology":null},{"url":null,"slug":"top-down-activity-representation-learning-for","title":"Top-down Activity Representation Learning for Video Question Answering","date":"2024-09-12","arxiv_id":"2409.07748","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-modality-bias-in-video-question","title":"Assessing Modality Bias in Video Question Answering Benchmarks with Multimodal Large Language Models","date":"2024-08-22","arxiv_id":"2408.12763","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-perception-benchmark","title":"Continuous Perception Benchmark","date":"2024-08-15","arxiv_id":"2408.07867","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-surg-towards-multimodal-surgical","title":"LLaVA-Surg: Towards Multimodal Surgical Assistant via Structured Surgical Video Learning","date":"2024-08-15","arxiv_id":"2408.07981","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-understanding-for-video-question","title":"Causal Understanding For Video Question Answering","date":"2024-07-23","arxiv_id":"2407.20257","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-video-question-answering-with","title":"End-to-End Video Question Answering with Frame Scoring Mechanisms and Adaptive Sampling","date":"2024-07-21","arxiv_id":"2407.15047","repositories_listed":0,"syntology":null},{"url":null,"slug":"vdma-video-question-answering-with","title":"VDMA: Video Question Answering with Dynamically Generated Multi-Agents","date":"2024-07-04","arxiv_id":"2407.03610","repositories_listed":0,"syntology":null},{"url":null,"slug":"align-and-aggregate-compositional-reasoning-1","title":"Align and Aggregate: Compositional Reasoning with Video Alignment and Answer Aggregation for Video Question-Answering","date":"2024-07-03","arxiv_id":"2407.03008","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyvideollm-towards-large-scale-video","title":"KeyVideoLLM: Towards Large-scale Video Keyframe Selection","date":"2024-07-03","arxiv_id":"2407.03104","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-solution-for-the-iccv-2023-perception","title":"The Solution for the ICCV 2023 Perception Test Challenge 2023 -- Task 6 -- Grounded videoQA","date":"2024-07-02","arxiv_id":"2407.01907","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-memory-for-long-video-qa","title":"Hierarchical Memory for Long Video QA","date":"2024-06-30","arxiv_id":"2407.00603","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-long-form-video-understanding","title":"Zero-Shot Long-Form Video Understanding through Screenplay","date":"2024-06-25","arxiv_id":"2406.17309","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-mitigation-prompts-long-term","title":"Hallucination Mitigation Prompts Long-term Video Understanding","date":"2024-06-17","arxiv_id":"2406.11333","repositories_listed":0,"syntology":null},{"url":"/paper/videollm-online-online-video-large-language-1","slug":"videollm-online-online-video-large-language-1","title":"VideoLLM-online: Online Video Large Language Model for Streaming Video","date":"2024-06-17","arxiv_id":"2406.11816","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-question-answering-for-people-with","title":"Video Question Answering for People with Visual Impairments Using an Egocentric 360-Degree Camera","date":"2024-05-30","arxiv_id":"2405.19794","repositories_listed":0,"syntology":null},{"url":null,"slug":"backpropagation-free-multi-modal-on-device","title":"Backpropagation-Free Multi-modal On-Device Model Adaptation via Cloud-Device Collaboration","date":"2024-05-21","arxiv_id":"2406.01601","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoqa-sc-adaptive-semantic-communication","title":"VideoQA-SC: Adaptive Semantic Communication for Video Question Answering","date":"2024-05-17","arxiv_id":"2406.18538","repositories_listed":0,"syntology":null},{"url":"/paper/cinepile-a-long-video-question-answering","slug":"cinepile-a-long-video-question-answering","title":"CinePile: A Long Video Question Answering Dataset and Benchmark","date":"2024-05-14","arxiv_id":"2405.08813","repositories_listed":0,"syntology":null},{"url":"/paper/capabilities-of-gemini-models-in-medicine","slug":"capabilities-of-gemini-models-in-medicine","title":"Capabilities of Gemini Models in Medicine","date":"2024-04-29","arxiv_id":"2404.18416","repositories_listed":0,"syntology":null},{"url":null,"slug":"pegasus-v1-technical-report","title":"Pegasus-v1 Technical Report","date":"2024-04-23","arxiv_id":"2404.14687","repositories_listed":0,"syntology":null},{"url":null,"slug":"reka-core-flash-and-edge-a-series-of-powerful","title":"Reka Core, Flash, and Edge: A Series of Powerful Multimodal Language Models","date":"2024-04-18","arxiv_id":"2404.12387","repositories_listed":0,"syntology":null},{"url":"/paper/morevqa-exploring-modular-reasoning-models","slug":"morevqa-exploring-modular-reasoning-models","title":"MoReVQA: Exploring Modular Reasoning Models for Video Question Answering","date":"2024-04-09","arxiv_id":"2404.06511","repositories_listed":0,"syntology":null},{"url":null,"slug":"koala-key-frame-conditioned-long-video-llm","title":"Koala: Key frame-conditioned long video-LLM","date":"2024-04-05","arxiv_id":"2404.04346","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-symbolic-videoqa-learning","title":"Neural-Symbolic VideoQA: Learning Compositional Spatio-Temporal Reasoning for Real-world Video Question Answering","date":"2024-04-05","arxiv_id":"2404.04007","repositories_listed":0,"syntology":null},{"url":null,"slug":"videodistill-language-aware-vision","title":"VideoDistill: Language-aware Vision Distillation for Video Question Answering","date":"2024-04-01","arxiv_id":"2404.00973","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranking-distillation-for-open-ended-video","title":"Ranking Distillation for Open-Ended Video Question Answering with Insufficient Labels","date":"2024-03-21","arxiv_id":"2403.14430","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-meet-long-video-advancing-long-video","title":"LLMs Meet Long Video: Advancing Long Video Question Answering with An Interactive Visual Adapter in LLMs","date":"2024-02-21","arxiv_id":"2402.13546","repositories_listed":0,"syntology":null},{"url":null,"slug":"slot-vlm-slowfast-slots-for-video-language","title":"Slot-VLM: SlowFast Slots for Video-Language Modeling","date":"2024-02-20","arxiv_id":"2402.13088","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoprism-a-foundational-visual-encoder-for","title":"VideoPrism: A Foundational Visual Encoder for Video Understanding","date":"2024-02-20","arxiv_id":"2402.13217","repositories_listed":0,"syntology":null},{"url":null,"slug":"bdiqa-a-new-dataset-for-video-question","title":"BDIQA: A New Dataset for Video Question Answering to Explore Cognitive Reasoning through Theory of Mind","date":"2024-02-12","arxiv_id":"2402.07402","repositories_listed":0,"syntology":null},{"url":null,"slug":"answering-from-sure-to-uncertain-uncertainty","title":"Answering from Sure to Uncertain: Uncertainty-Aware Curriculum Learning for Video Question Answering","date":"2024-01-03","arxiv_id":"2401.01510","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-aware-visual-semantic-distillation","title":"Language-aware Visual Semantic Distillation for Video Question Answering","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-scaling-up-a-multilingual-vision-and","title":"On Scaling Up a Multilingual Vision and Language Model","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vista-llama-reducing-hallucination-in-video","title":"VISTA-LLAMA: Reducing Hallucination in Video Language Models via Equal Distance to Visual Tokens","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-reasoning-with-event-correlation","title":"Cross-Modal Reasoning with Event Correlation for Video Question Answering","date":"2023-12-20","arxiv_id":"2312.12721","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-test-2023-a-summary-of-the-first","title":"Perception Test 2023: A Summary of the First Challenge And Outcome","date":"2023-12-20","arxiv_id":"2312.13090","repositories_listed":0,"syntology":null},{"url":"/paper/text-conditioned-resampler-for-long-form","slug":"text-conditioned-resampler-for-long-form","title":"Text-Conditioned Resampler For Long Form Video Understanding","date":"2023-12-19","arxiv_id":"2312.11897","repositories_listed":0,"syntology":null},{"url":"/paper/vista-llama-reliable-video-narrator-via-equal","slug":"vista-llama-reliable-video-narrator-via-equal","title":"Vista-LLaMA: Reliable Video Narrator via Equal Distance to Visual Tokens","date":"2023-12-12","arxiv_id":"2312.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"movqa-a-benchmark-of-versatile-question","title":"MoVQA: A Benchmark of Versatile Question-Answering for Long-Form Movie Understanding","date":"2023-12-08","arxiv_id":"2312.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-based-video-language-model-for","title":"Retrieval-based Video Language Model for Efficient Long Video Question Answering","date":"2023-12-08","arxiv_id":"2312.04931","repositories_listed":0,"syntology":null},{"url":null,"slug":"vaquita-enhancing-alignment-in-llm-assisted","title":"VaQuitA: Enhancing Alignment in LLM-Assisted Video Understanding","date":"2023-12-04","arxiv_id":"2312.02310","repositories_listed":0,"syntology":null},{"url":"/paper/zero-shot-video-question-answering-with","slug":"zero-shot-video-question-answering-with","title":"Zero-Shot Video Question Answering with Procedural Programs","date":"2023-12-01","arxiv_id":"2312.00937","repositories_listed":0,"syntology":null},{"url":null,"slug":"e-vilm-efficient-video-language-model-via","title":"E-ViLM: Efficient Video-Language Model via Masked Video Modeling with Semantic Vector-Quantized Tokenizer","date":"2023-11-28","arxiv_id":"2311.17267","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-video-question-answering-with","title":"Characterizing Video Question Answering with Sparsified Inputs","date":"2023-11-27","arxiv_id":"2311.16311","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt4video-a-unified-multimodal-large-language","title":"GPT4Video: A Unified Multimodal Large Language Model for lnstruction-Followed Understanding and Safety-Aware Generation","date":"2023-11-25","arxiv_id":"2311.16511","repositories_listed":0,"syntology":null},{"url":"/paper/mirasol3b-a-multimodal-autoregressive-model","slug":"mirasol3b-a-multimodal-autoregressive-model","title":"Mirasol3B: A Multimodal Autoregressive model for time-aligned and contextual modalities","date":"2023-11-09","arxiv_id":"2311.05698","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-blended-attention-network-for-video","title":"Modular Blended Attention Network for Video Question Answering","date":"2023-11-02","arxiv_id":"2311.12866","repositories_listed":0,"syntology":null},{"url":"/paper/mmtf-multi-modal-temporal-fusion-for","slug":"mmtf-multi-modal-temporal-fusion-for","title":"MMTF: Multi-Modal Temporal Fusion for Commonsense Video Question Answering","date":"2023-10-06","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/atm-action-temporality-modeling-for-video","slug":"atm-action-temporality-modeling-for-video","title":"ATM: Action Temporality Modeling for Video Question Answering","date":"2023-09-05","arxiv_id":"2309.02290","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-video-scenes-through-text","title":"Understanding Video Scenes through Text: Insights from Text-based Video Question Answering","date":"2023-09-04","arxiv_id":"2309.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"distraction-free-embeddings-for-robust-vqa","title":"Distraction-free Embeddings for Robust VQA","date":"2023-08-31","arxiv_id":"2309.00133","repositories_listed":0,"syntology":null},{"url":null,"slug":"redundancy-aware-transformer-for-video","title":"Redundancy-aware Transformer for Video Question Answering","date":"2023-08-07","arxiv_id":"2308.03267","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-aware-relative-spatio-temporal-graph","title":"Keyword-Aware Relative Spatio-Temporal Graph Networks for Video Question Answering","date":"2023-07-25","arxiv_id":"2307.13250","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-domain-video-question-answering-with","title":"Traffic-Domain Video Question Answering with Automatic Captioning","date":"2023-07-18","arxiv_id":"2307.09636","repositories_listed":0,"syntology":null},{"url":null,"slug":"read-look-or-listen-what-s-needed-for-solving","title":"Read, Look or Listen? What's Needed for Solving a Multimodal Dataset","date":"2023-07-06","arxiv_id":"2307.04532","repositories_listed":0,"syntology":null},{"url":"/paper/retrieving-to-answer-zero-shot-video-question","slug":"retrieving-to-answer-zero-shot-video-question","title":"Retrieving-to-Answer: Zero-Shot Video Question Answering with Frozen Large Language Models","date":"2023-06-15","arxiv_id":"2306.11732","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversifying-joint-vision-language","title":"Diversifying Joint Vision-Language Tokenization Learning","date":"2023-06-06","arxiv_id":"2306.03421","repositories_listed":0,"syntology":null},{"url":"/paper/vlab-enhancing-video-language-pre-training-by","slug":"vlab-enhancing-video-language-pre-training-by","title":"VLAB: Enhancing Video Language Pre-training by Feature Adapting and Blending","date":"2023-05-22","arxiv_id":"2305.13167","repositories_listed":0,"syntology":null},{"url":null,"slug":"tg-vqa-ternary-game-of-video-question","title":"TG-VQA: Ternary Game of Video Question Answering","date":"2023-05-17","arxiv_id":"2305.10049","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-a-video-worth-n-times-n-images-a-highly","title":"Is a Video worth $n\\times n$ Images? A Highly Efficient Approach to Transformer-based Video Question Answering","date":"2023-05-16","arxiv_id":"2305.09107","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-dynamic-retrospective","title":"Semantic-aware Dynamic Retrospective-Prospective Reasoning for Event-level Video Question Answering","date":"2023-05-14","arxiv_id":"2305.08059","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoofa-two-stage-pre-training-for-video-to","title":"VideoOFA: Two-Stage Pre-Training for Video-to-Text Generation","date":"2023-05-04","arxiv_id":"2305.03204","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-learning-for-video","title":"A Review of Deep Learning for Video Captioning","date":"2023-04-22","arxiv_id":"2304.11431","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-are-causal-knowledge","title":"Language Models are Causal Knowledge Extractors for Zero-shot Video Question Answering","date":"2023-04-07","arxiv_id":"2304.03754","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporally-discriminative-video","title":"Structured Video-Language Modeling with Temporal Grouping and Spatial Grounding","date":"2023-03-28","arxiv_id":"2303.16341","repositories_listed":0,"syntology":null},{"url":"/paper/multi-efficient-video-and-language","slug":"multi-efficient-video-and-language","title":"MuLTI: Efficient Video-and-Language Understanding with Text-Guided MultiWay-Sampler and Multiple Choice Modeling","date":"2023-03-10","arxiv_id":"2303.05707","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-question-answering-using-clip-guided","title":"Video Question Answering Using CLIP-Guided Visual-Text Attention","date":"2023-03-06","arxiv_id":"2303.03131","repositories_listed":0,"syntology":null},{"url":null,"slug":"stoa-vlp-spatial-temporal-modeling-of-object","title":"STOA-VLP: Spatial-Temporal Modeling of Object and Action for Video-Language Pre-training","date":"2023-02-20","arxiv_id":"2302.09736","repositories_listed":0,"syntology":null},{"url":"/paper/semi-parametric-video-grounded-text","slug":"semi-parametric-video-grounded-text","title":"Semi-Parametric Video-Grounded Text Generation","date":"2023-01-27","arxiv_id":"2301.11507","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-perceiving-video-language-pre","title":"Temporal Perceiving Video-Language Pre-training","date":"2023-01-18","arxiv_id":"2301.07463","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-trajectory-word-alignments-for-video","title":"Learning Trajectory-Word Alignments for Video-Language Tasks","date":"2023-01-05","arxiv_id":"2301.01953","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-the-real-association-multimodal","title":"Discovering the Real Association: Multimodal Causal Reasoning in Video Question Answering","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-proxy-intervention-for-deconfounded","title":"Knowledge Proxy Intervention for Deconfounded Video Question Answering","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/hitea-hierarchical-temporal-aware-video","slug":"hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","arxiv_id":"2212.14546","repositories_listed":0,"syntology":null},{"url":"/paper/video-text-modeling-with-zero-shot-transfer","slug":"video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","arxiv_id":"2212.04979","repositories_listed":0,"syntology":null},{"url":null,"slug":"smaug-sparse-masked-autoencoder-for-efficient","title":"SMAUG: Sparse Masked Autoencoder for Efficient Video-Language Pre-training","date":"2022-11-21","arxiv_id":"2211.11446","repositories_listed":0,"syntology":null},{"url":null,"slug":"watching-the-news-towards-videoqa-models-that","title":"Watching the News: Towards VideoQA Models that can Read","date":"2022-11-10","arxiv_id":"2211.05588","repositories_listed":0,"syntology":null},{"url":null,"slug":"litevl-efficient-video-language-learning-with","title":"LiteVL: Efficient Video-Language Learning with Enhanced Spatial-Temporal Modeling","date":"2022-10-21","arxiv_id":"2210.11929","repositories_listed":0,"syntology":null},{"url":"/paper/composing-ensembles-of-pre-trained-models-via","slug":"composing-ensembles-of-pre-trained-models-via","title":"Composing Ensembles of Pre-trained Models via Iterative Consensus","date":"2022-10-20","arxiv_id":"2210.11522","repositories_listed":0,"syntology":null},{"url":null,"slug":"dense-but-efficient-videoqa-for-intricate","title":"Dense but Efficient VideoQA for Intricate Compositional Reasoning","date":"2022-10-19","arxiv_id":"2210.10300","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-video-language-learning-with-fine","title":"Contrastive Video-Language Learning with Fine-grained Frame Sampling","date":"2022-10-10","arxiv_id":"2210.05039","repositories_listed":0,"syntology":null},{"url":null,"slug":"locate-before-answering-answer-guided","title":"Locate before Answering: Answer Guided Question Localization for Video Question Answering","date":"2022-10-05","arxiv_id":"2210.02081","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-the-wild-video-question-answering","title":"In-the-Wild Video Question Answering","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/omnivl-one-foundation-model-for-image","slug":"omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","arxiv_id":"2209.07526","repositories_listed":0,"syntology":null},{"url":"/paper/wildqa-in-the-wild-video-question-answering","slug":"wildqa-in-the-wild-video-question-answering","title":"WildQA: In-the-Wild Video Question Answering","date":"2022-09-14","arxiv_id":"2209.06650","repositories_listed":0,"syntology":null},{"url":null,"slug":"frame-subtitle-self-supervision-for-multi","title":"Frame-Subtitle Self-Supervision for Multi-Modal Video Question Answering","date":"2022-09-08","arxiv_id":"2209.03609","repositories_listed":0,"syntology":null},{"url":"/paper/video-question-answering-with-iterative-video","slug":"video-question-answering-with-iterative-video","title":"Video Question Answering with Iterative Video-Text Co-Tokenization","date":"2022-08-01","arxiv_id":"2208.00934","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-multistep-reasoning-based-on-video","title":"Dynamic Multistep Reasoning based on Video Scene Graph for Video Question Answering","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/0-1-deep-neural-networks-via-block-coordinate","slug":"0-1-deep-neural-networks-via-block-coordinate","title":"0/1 Deep Neural Networks via Block Coordinate Descent","date":"2022-06-19","arxiv_id":"2206.09379","repositories_listed":0,"syntology":null}],"record_sha256":"9225b0673d1e9519e3be68788356d0c9afe98164053fd7279507424a80a345eb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}