{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-question-answering/papers/5","list_of":"/task/video-question-answering","task":"Video Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":5,"rows_per_page":100,"rows":[401,460],"of":460,"counts":{"archive_papers_tagged":460,"with_a_code_link":250,"where_syntology_ran_a_sample":124,"not_listed_spam_title":0,"listed":460,"listed_where_code_ran":124,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":107,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":107,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-question-answering","prev":"/task/video-question-answering/papers/4","next":null,"papers":[{"url":null,"slug":"structured-two-stream-attention-network-for","title":"Structured Two-stream Attention Network for Video Question Answering","date":"2022-06-02","arxiv_id":"2206.01017","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-alignment-between-deep","title":"Modality Alignment between Deep Representations for Effective Video-and-Language Learning","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-semantic-composition-with-syntactic","title":"Modeling Semantic Composition with Syntactic Hypergraph for Video Question Answering","date":"2022-05-13","arxiv_id":"2205.06530","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-the-medvidqa-2022-shared-task-on","title":"Overview of the MedVidQA 2022 Shared Task on Medical Video Question-Answering","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-language-co-attention-with-multimodal","title":"Video Language Co-Attention with Multimodal Fast-Learning Feature Fusion for VideoQA","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-multi-modal-alignment-in-video","title":"Rethinking Multi-Modal Alignment in Video Question Answering from Feature and Sample Perspectives","date":"2022-04-25","arxiv_id":"2204.11544","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-compositional-consistency-for-video","title":"Measuring Compositional Consistency for Video Question Answering","date":"2022-04-14","arxiv_id":"2204.07190","repositories_listed":0,"syntology":null},{"url":"/paper/2-5-1-d-spatio-temporal-scene-graphs-for","slug":"2-5-1-d-spatio-temporal-scene-graphs-for","title":"(2.5+1)D Spatio-Temporal Scene Graphs for Video Question Answering","date":"2022-02-18","arxiv_id":"2202.09277","repositories_listed":0,"syntology":null},{"url":"/paper/newskvqa-knowledge-aware-news-video-question","slug":"newskvqa-knowledge-aware-news-video-question","title":"NEWSKVQA: Knowledge-Aware News Video Question Answering","date":"2022-02-08","arxiv_id":"2202.04015","repositories_listed":0,"syntology":null},{"url":null,"slug":"coco-bert-improving-video-language-pre","title":"CoCo-BERT: Improving Video-Language Pre-training with Contrastive Cross-modal Matching and Denoising","date":"2021-12-14","arxiv_id":"2112.07515","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-inside-self-driven-siamese","title":"Learning from Inside: Self-driven Siamese Sampling and Reasoning for Video Question Answering","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"livlr-a-lightweight-visual-linguistic","title":"LiVLR: A Lightweight Visual-Linguistic Reasoning Framework for Video Question Answering","date":"2021-11-29","arxiv_id":"2111.14547","repositories_listed":0,"syntology":null},{"url":null,"slug":"craft-a-benchmark-for-causal-reasoning-about-1","title":"CRAFT: A Benchmark for Causal Reasoning About Forces and inTeractions","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fill-in-the-blank-a-challenging-video","title":"Fill-in-the-Blank: A Challenging Video Understanding Evaluation Framework","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-domain-agnostic-knowledge-in","title":"Transferring Domain-Agnostic Knowledge in Video Question Answering","date":"2021-10-26","arxiv_id":"2110.13395","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multi-modal-video-reasoning-and-analyzing","title":"The Multi-Modal Video Reasoning and Analyzing Competition","date":"2021-08-18","arxiv_id":"2108.08344","repositories_listed":0,"syntology":null},{"url":null,"slug":"mounting-video-metadata-on-transformer-based","title":"Mounting Video Metadata on Transformer-based Language Model for Open-ended Video Question Answering","date":"2021-08-11","arxiv_id":"2108.05158","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-progressive-attention-network-for","title":"Multi-Scale Progressive Attention Network for Video Question Answering","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cogme-a-novel-evaluation-metric-for-video","title":"CogME: A Cognition-Inspired Multi-Dimensional Evaluation Metric for Story Understanding","date":"2021-07-21","arxiv_id":"2107.09847","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-object-oriented-spatio-temporal","title":"Hierarchical Object-oriented Spatio-Temporal Reasoning for Video Question Answering","date":"2021-06-25","arxiv_id":"2106.13432","repositories_listed":0,"syntology":null},{"url":null,"slug":"ireason-multimodal-commonsense-reasoning","title":"iReason: Multimodal Commonsense Reasoning using Videos and Natural Language with Interpretability","date":"2021-06-25","arxiv_id":"2107.10300","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-rehearse-in-long-sequence","title":"Learning to Rehearse in Long Sequence Memorization","date":"2021-06-02","arxiv_id":"2106.01096","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridge-to-answer-structure-aware-graph","title":"Bridge to Answer: Structure-aware Graph Interaction Network for Video Question Answering","date":"2021-04-29","arxiv_id":"2104.14085","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-centric-representation-learning-for","title":"Object-Centric Representation Learning for Video Question Answering","date":"2021-04-12","arxiv_id":"2104.05166","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-question-answering-with-phrases-via","title":"Video Question Answering with Phrases via Semantic Roles","date":"2021-04-08","arxiv_id":"2104.03762","repositories_listed":0,"syntology":null},{"url":null,"slug":"cupid-adaptive-curation-of-pre-training-data","title":"CUPID: Adaptive Curation of Pre-training Data for Video-and-Language Representation Learning","date":"2021-04-01","arxiv_id":"2104.00285","repositories_listed":0,"syntology":null},{"url":"/paper/agqa-a-benchmark-for-compositional-spatio","slug":"agqa-a-benchmark-for-compositional-spatio","title":"AGQA: A Benchmark for Compositional Spatio-Temporal Reasoning","date":"2021-03-30","arxiv_id":"2103.16002","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyster-a-hybrid-spatio-temporal-event","title":"HySTER: A Hybrid Spatio-Temporal Event Reasoner","date":"2021-01-17","arxiv_id":"2101.06644","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advances-in-video-question-answering-a","title":"Recent Advances in Video Question Answering: A Review of Datasets and Methods","date":"2021-01-15","arxiv_id":"2101.05954","repositories_listed":0,"syntology":null},{"url":null,"slug":"env-qa-a-video-question-answering-benchmark","title":"Env-QA: A Video Question Answering Benchmark for Comprehensive Understanding of Dynamic Environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hair-hierarchical-visual-semantic-relational","title":"HAIR: Hierarchical Visual-Semantic Relational Reasoning for Video Question Answering","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-question-answering-using-language","title":"Video Question Answering Using Language-Guided Deep Compressed-Domain Video Feature","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"trying-bilinear-pooling-in-video-qa","title":"Trying Bilinear Pooling in Video-QA","date":"2020-12-18","arxiv_id":"2012.10285","repositories_listed":0,"syntology":null},{"url":"/paper/iperceive-applying-common-sense-reasoning-to-1","slug":"iperceive-applying-common-sense-reasoning-to-1","title":"iPerceive: Applying Common-Sense Reasoning to Multi-Modal Dense Video Captioning and Video Question Answering","date":"2020-11-16","arxiv_id":"2011.07735","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-attentional-transformers-for-story-based","title":"Co-attentional Transformers for Story-Based Video Understanding","date":"2020-10-27","arxiv_id":"2010.14104","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-conditional-relation-networks-1","title":"Hierarchical Conditional Relation Networks for Multimodal Video Question Answering","date":"2020-10-18","arxiv_id":"2010.10019","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-pre-training-and-contrastive","title":"Self-supervised pre-training and contrastive representation learning for multiple-choice video QA","date":"2020-09-17","arxiv_id":"2009.08043","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-techniques-for-the-video","title":"Data augmentation techniques for the Video Question Answering task","date":"2020-08-22","arxiv_id":"2008.09849","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-question-answering-on-screencast","title":"Video Question Answering on Screencast Tutorials","date":"2020-08-02","arxiv_id":"2008.00544","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-gives-the-answer-away-question-answering","title":"What Gives the Answer Away? Question Answering Bias Analysis on Video QA Datasets","date":"2020-07-07","arxiv_id":"2007.03626","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-captions-on-gif-a-large-scale-video","title":"Auto-captions on GIF: A Large-scale Video-sentence Dataset for Vision-language Pre-training","date":"2020-07-05","arxiv_id":"2007.02375","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-shifting-attention-network-for-multi-1","title":"Modality Shifting Attention Network for Multi-modal Video Question Answering","date":"2020-07-04","arxiv_id":"2007.02036","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-based-visual-question-answering-in","title":"Knowledge-Based Visual Question Answering in Videos","date":"2020-04-17","arxiv_id":"2004.08385","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-transformer-with-pointer-network","title":"Multimodal Transformer with Pointer Network for the DSTC8 AVSD Challenge","date":"2020-02-25","arxiv_id":"2002.10695","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-dialog-via-progressive-inference-and","title":"Video Dialog via Progressive Inference and Cross-Transformer","date":"2019-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/knowit-vqa-answering-knowledge-based","slug":"knowit-vqa-answering-knowledge-based","title":"KnowIT VQA: Answering Knowledge-Based Questions about Videos","date":"2019-10-23","arxiv_id":"1910.10706","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-question-guided-video-representation","title":"Learning Question-Guided Video Representation for Multi-Turn Video Question Answering","date":"2019-07-31","arxiv_id":"1907.13280","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason-with-relational-video","title":"Neural Reasoning, Fast and Slow, for Video Question Answering","date":"2019-07-10","arxiv_id":"1907.04553","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-question-generation-via-cross-modal","title":"Video Question Generation via Cross-Modal Self-Attention Networks Learning","date":"2019-07-05","arxiv_id":"1907.03049","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-ended-long-form-video-question-answering","title":"Open-Ended Long-Form Video Question Answering via Hierarchical Convolutional Self-Attention Networks","date":"2019-06-28","arxiv_id":"1906.12158","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-multimodal-network-for-movie","title":"Adversarial Multimodal Network for Movie Question Answering","date":"2019-06-24","arxiv_id":"1906.09844","repositories_listed":0,"syntology":null},{"url":null,"slug":"190513540","title":"Gaining Extra Supervision via Multi-task learning for Multi-Modal Video Question Answering","date":"2019-05-28","arxiv_id":"1905.13540","repositories_listed":0,"syntology":null},{"url":null,"slug":"holistic-multi-modal-memory-network-for-movie","title":"Holistic Multi-modal Memory Network for Movie Question Answering","date":"2018-11-12","arxiv_id":"1811.04595","repositories_listed":0,"syntology":null},{"url":null,"slug":"movie-question-answering-remembering-the","title":"Movie Question Answering: Remembering the Textual Cues for Layered Visual Contents","date":"2018-04-25","arxiv_id":"1804.09412","repositories_listed":0,"syntology":null},{"url":"/paper/motion-appearance-co-memory-networks-for","slug":"motion-appearance-co-memory-networks-for","title":"Motion-Appearance Co-Memory Networks for Video Question Answering","date":"2018-03-29","arxiv_id":"1803.10906","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-question-answering-via-attribute","title":"Video Question Answering via Attribute-Augmented Attention Network Learning","date":"2017-07-20","arxiv_id":"1707.06355","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-forgettable-watcher-model-for-video","title":"The Forgettable-Watcher Model for Video Question Answering","date":"2017-05-03","arxiv_id":"1705.01253","repositories_listed":0,"syntology":null},{"url":null,"slug":"marioqa-answering-questions-by-watching","title":"MarioQA: Answering Questions by Watching Gameplay Videos","date":"2016-12-06","arxiv_id":"1612.01669","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-video-descriptions-to-learn-video","title":"Leveraging Video Descriptions to Learn Video Question Answering","date":"2016-11-12","arxiv_id":"1611.04021","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncovering-temporal-context-for-video","title":"Uncovering Temporal Context for Video Question and Answering","date":"2015-11-15","arxiv_id":"1511.04670","repositories_listed":0,"syntology":null}],"record_sha256":"fb185ac199f84988ee979a17c3ab67760e1985fdcbe46759b5b49aff894071b5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}