{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-question-answering/papers/3","list_of":"/task/video-question-answering","task":"Video Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":5,"rows_per_page":100,"rows":[201,300],"of":460,"counts":{"archive_papers_tagged":460,"with_a_code_link":250,"where_syntology_ran_a_sample":124,"not_listed_spam_title":0,"listed":460,"listed_where_code_ran":124,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":107,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":107,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-question-answering","prev":"/task/video-question-answering/papers/2","next":"/task/video-question-answering/papers/4","papers":[{"url":"/paper/learning-fine-grained-visual-understanding","slug":"learning-fine-grained-visual-understanding","title":"Learning Fine-Grained Visual Understanding for Video Question Answering via Decoupling Spatial-Temporal Modeling","date":"2022-10-08","arxiv_id":"2210.03941","repositories_listed":1,"syntology":null},{"url":"/paper/extending-compositional-attention-networks","slug":"extending-compositional-attention-networks","title":"Extending Compositional Attention Networks for Social Reasoning in Videos","date":"2022-10-03","arxiv_id":"2210.01191","repositories_listed":1,"syntology":null},{"url":"/paper/lavis-a-library-for-language-vision","slug":"lavis-a-library-for-language-vision","title":"LAVIS: A Library for Language-Vision Intelligence","date":"2022-09-15","arxiv_id":"2209.09019","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-end-to-end-video","slug":"an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","arxiv_id":"2209.01540","repositories_listed":1,"syntology":null},{"url":"/paper/equivariant-and-invariant-grounding-for-video","slug":"equivariant-and-invariant-grounding-for-video","title":"Equivariant and Invariant Grounding for Video Question Answering","date":"2022-07-26","arxiv_id":"2207.12783","repositories_listed":1,"syntology":null},{"url":"/paper/clover-towards-a-unified-video-language","slug":"clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","arxiv_id":"2207.07885","repositories_listed":1,"syntology":null},{"url":"/paper/video-graph-transformer-for-video-question","slug":"video-graph-transformer-for-video-question","title":"Video Graph Transformer for Video Question Answering","date":"2022-07-12","arxiv_id":"2207.05342","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":5,"n_ran_checked":8,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/video-graph-transformer-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2207.05342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05342"}},"official":{"repos":["sail-sg/vgt"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/video-dialog-as-conversation-about-objects","slug":"video-dialog-as-conversation-about-objects","title":"Video Dialog as Conversation about Objects Living in Space-Time","date":"2022-07-08","arxiv_id":"2207.03656","repositories_listed":1,"syntology":null},{"url":"/paper/lavender-unifying-video-language","slug":"lavender-unifying-video-language","title":"LAVENDER: Unifying Video-Language Understanding as Masked Language Modeling","date":"2022-06-14","arxiv_id":"2206.07160","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/lavender-unifying-video-language#ran","syntology_url":"https://syntology.ai/paper/2206.07160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07160"}},"official":{"repos":["microsoft/lavender"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/invariant-grounding-for-video-question-1","slug":"invariant-grounding-for-video-question-1","title":"Invariant Grounding for Video Question Answering","date":"2022-06-06","arxiv_id":"2206.02349","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/invariant-grounding-for-video-question-1#ran","syntology_url":"https://syntology.ai/paper/2206.02349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02349"}},"official":{"repos":["yl3800/igv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-fast-adaptation-of-pretrained","slug":"towards-fast-adaptation-of-pretrained","title":"Towards Fast Adaptation of Pretrained Contrastive Models for Multi-channel Video-Language Retrieval","date":"2022-06-05","arxiv_id":"2206.02082","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-fast-adaptation-of-pretrained#ran","syntology_url":"https://syntology.ai/paper/2206.02082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02082"}},"official":{"repos":["xudonglinthu/upgradable-multimodal-intelligence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-video-in-video-language","slug":"revisiting-the-video-in-video-language","title":"Revisiting the \"Video\" in Video-Language Understanding","date":"2022-06-03","arxiv_id":"2206.01720","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/revisiting-the-video-in-video-language#ran","syntology_url":"https://syntology.ai/paper/2206.01720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01720"}},"official":null}},{"url":"/paper/from-representation-to-reasoning-towards-both","slug":"from-representation-to-reasoning-towards-both","title":"From Representation to Reasoning: Towards both Evidence and Commonsense Reasoning for Video Question-Answering","date":"2022-05-30","arxiv_id":"2205.14895","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-representation-to-reasoning-towards-both#ran","syntology_url":"https://syntology.ai/paper/2205.14895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14895"}},"official":{"repos":["bcmi/causal-vidqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-with-image-descriptors-are","slug":"language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","arxiv_id":"2205.10747","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-with-image-descriptors-are#ran","syntology_url":"https://syntology.ai/paper/2205.10747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10747"}},"official":{"repos":["mikewangwzhl/vidil"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-answer-visual-questions-from-web","slug":"learning-to-answer-visual-questions-from-web","title":"Learning to Answer Visual Questions from Web Videos","date":"2022-05-10","arxiv_id":"2205.05019","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-to-answer-visual-questions-from-web#ran","syntology_url":"https://syntology.ai/paper/2205.05019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05019"}},"official":{"repos":["antoyang/just-ask"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/multilevel-hierarchical-network-with","slug":"multilevel-hierarchical-network-with","title":"Multilevel Hierarchical Network with Multiscale Sampling for Video Question Answering","date":"2022-05-09","arxiv_id":"2205.04061","repositories_listed":1,"syntology":null},{"url":"/paper/all-in-one-exploring-unified-video-language","slug":"all-in-one-exploring-unified-video-language","title":"All in One: Exploring Unified Video-Language Pre-training","date":"2022-03-14","arxiv_id":"2203.07303","repositories_listed":1,"syntology":null},{"url":"/paper/video-question-answering-datasets-algorithms","slug":"video-question-answering-datasets-algorithms","title":"Video Question Answering: Datasets, Algorithms and Challenges","date":"2022-03-02","arxiv_id":"2203.01225","repositories_listed":1,"syntology":null},{"url":"/paper/video-as-conditional-graph-hierarchy-for","slug":"video-as-conditional-graph-hierarchy-for","title":"Video as Conditional Graph Hierarchy for Multi-Granular Question Answering","date":"2021-12-12","arxiv_id":"2112.06197","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-as-conditional-graph-hierarchy-for#ran","syntology_url":"https://syntology.ai/paper/2112.06197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06197"}},"official":{"repos":["doc-doc/hqga"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/swinbert-end-to-end-transformers-with-sparse","slug":"swinbert-end-to-end-transformers-with-sparse","title":"SwinBERT: End-to-End Transformers with Sparse Attention for Video Captioning","date":"2021-11-25","arxiv_id":"2111.13196","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swinbert-end-to-end-transformers-with-sparse#ran","syntology_url":"https://syntology.ai/paper/2111.13196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13196"}},"official":{"repos":["microsoft/swinbert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/violet-end-to-end-video-language-transformers","slug":"violet-end-to-end-video-language-transformers","title":"VIOLET : End-to-End Video-Language Transformers with Masked Visual-token Modeling","date":"2021-11-24","arxiv_id":"2111.12681","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/violet-end-to-end-video-language-transformers#ran","syntology_url":"https://syntology.ai/paper/2111.12681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12681"}},"official":{"repos":["tsujuifu/pytorch_violet"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/temporal-pyramid-transformer-with-multimodal","slug":"temporal-pyramid-transformer-with-multimodal","title":"Temporal Pyramid Transformer with Multimodal Interaction for Video Question Answering","date":"2021-09-10","arxiv_id":"2109.04735","repositories_listed":1,"syntology":null},{"url":"/paper/dualvgr-a-dual-visual-graph-reasoning-unit","slug":"dualvgr-a-dual-visual-graph-reasoning-unit","title":"DualVGR: A Dual-Visual Graph Reasoning Unit for Video Question Answering","date":"2021-07-10","arxiv_id":"2107.04768","repositories_listed":1,"syntology":null},{"url":"/paper/attend-what-you-need-motion-appearance","slug":"attend-what-you-need-motion-appearance","title":"Attend What You Need: Motion-Appearance Synergistic Networks for Video Question Answering","date":"2021-06-19","arxiv_id":"2106.10446","repositories_listed":1,"syntology":{"n":20,"n_ran":15,"n_constructed":14,"n_ran_checked":15,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/attend-what-you-need-motion-appearance#ran","syntology_url":"https://syntology.ai/paper/2106.10446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.10446"}},"official":{"repos":["ahjeongseo/MASN-pytorch"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/next-qa-next-phase-of-question-answering-to-1","slug":"next-qa-next-phase-of-question-answering-to-1","title":"NExT-QA: Next Phase of Question-Answering to Explaining Temporal Actions","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/value-a-multi-task-benchmark-for-video-and","slug":"value-a-multi-task-benchmark-for-video-and","title":"VALUE: A Multi-Task Benchmark for Video-and-Language Understanding Evaluation","date":"2021-06-08","arxiv_id":"2106.04632","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/value-a-multi-task-benchmark-for-video-and#ran","syntology_url":"https://syntology.ai/paper/2106.04632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04632"}},"official":{"repos":["VALUE-Leaderboard/StarterCode"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/decembert-learning-from-noisy-instructional","slug":"decembert-learning-from-noisy-instructional","title":"DeCEMBERT: Learning from Noisy Instructional Videos via Dense Captions and Entropy Minimization","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/relation-aware-hierarchical-attention","slug":"relation-aware-hierarchical-attention","title":"Relation-aware Hierarchical Attention Framework for Video Question Answering","date":"2021-05-13","arxiv_id":"2105.06160","repositories_listed":1,"syntology":null},{"url":"/paper/fill-in-the-blank-as-a-challenging-video","slug":"fill-in-the-blank-as-a-challenging-video","title":"FIBER: Fill-in-the-Blanks as a Challenging Video Understanding Evaluation Framework","date":"2021-04-09","arxiv_id":"2104.04182","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-vision-and-language-from-the-video","slug":"bridging-vision-and-language-from-the-video","title":"A Comprehensive Review of the Video-to-Text Problem","date":"2021-03-27","arxiv_id":"2103.14785","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-hidden-treasure-of-dialog-in-video","slug":"on-the-hidden-treasure-of-dialog-in-video","title":"On the hidden treasure of dialog in video question answering","date":"2021-03-26","arxiv_id":"2103.14517","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-clipbert-for-video-and-language","slug":"less-is-more-clipbert-for-video-and-language","title":"Less is More: ClipBERT for Video-and-Language Learning via Sparse Sampling","date":"2021-02-11","arxiv_id":"2102.06183","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-video-question-answer-generation","slug":"end-to-end-video-question-answer-generation","title":"End-to-End Video Question-Answer Generation with Generator-Pretester Network","date":"2021-01-05","arxiv_id":"2101.01447","repositories_listed":1,"syntology":null},{"url":"/paper/on-modality-bias-in-the-tvqa-dataset","slug":"on-modality-bias-in-the-tvqa-dataset","title":"On Modality Bias in the TVQA Dataset","date":"2020-12-18","arxiv_id":"2012.10210","repositories_listed":1,"syntology":null},{"url":"/paper/craft-a-benchmark-for-causal-reasoning-about","slug":"craft-a-benchmark-for-causal-reasoning-about","title":"CRAFT: A Benchmark for Causal Reasoning About Forces and inTeractions","date":"2020-12-08","arxiv_id":"2012.04293","repositories_listed":1,"syntology":null},{"url":"/paper/just-ask-learning-to-answer-questions-from","slug":"just-ask-learning-to-answer-questions-from","title":"Just Ask: Learning to Answer Questions from Millions of Narrated Videos","date":"2020-12-01","arxiv_id":"2012.00451","repositories_listed":1,"syntology":null},{"url":"/paper/open-ended-multi-modal-relational-reason-for","slug":"open-ended-multi-modal-relational-reason-for","title":"Open-Ended Multi-Modal Relational Reasoning for Video Question Answering","date":"2020-12-01","arxiv_id":"2012.00822","repositories_listed":1,"syntology":null},{"url":"/paper/actbert-learning-global-local-video-text-1","slug":"actbert-learning-global-local-video-text-1","title":"ActBERT: Learning Global-Local Video-Text Representations","date":"2020-11-14","arxiv_id":"2011.07231","repositories_listed":1,"syntology":null},{"url":"/paper/location-aware-graph-convolutional-networks","slug":"location-aware-graph-convolutional-networks","title":"Location-aware Graph Convolutional Networks for Video Question Answering","date":"2020-08-07","arxiv_id":"2008.09105","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/location-aware-graph-convolutional-networks#ran","syntology_url":"https://syntology.ai/paper/2008.09105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09105"}},"official":{"repos":["SunDoge/L-GCN"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-based-video-question-answering-with","slug":"knowledge-based-video-question-answering-with","title":"Knowledge-Based Video Question Answering with Unsupervised Scene Descriptions","date":"2020-07-17","arxiv_id":"2007.08751","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-based-video-question-answering-with#ran","syntology_url":"https://syntology.ai/paper/2007.08751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.08751"}},"official":{"repos":["noagarcia/ROLL-VideoQA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-relation-grounding-in-videos","slug":"visual-relation-grounding-in-videos","title":"Visual Relation Grounding in Videos","date":"2020-07-17","arxiv_id":"2007.08814","repositories_listed":1,"syntology":null},{"url":"/paper/dense-caption-matching-and-frame-selection","slug":"dense-caption-matching-and-frame-selection","title":"Dense-Caption Matching and Frame-Selection Gating for Temporal Localization in VideoQA","date":"2020-05-13","arxiv_id":"2005.06409","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":6,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dense-caption-matching-and-frame-selection#ran","syntology_url":"https://syntology.ai/paper/2005.06409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.06409"}},"official":{"repos":["hyounghk/VideoQADenseCapFrameGate-ACL2020"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/dramaqa-character-centered-video-story","slug":"dramaqa-character-centered-video-story","title":"DramaQA: Character-Centered Video Story Understanding with Hierarchical QA","date":"2020-05-07","arxiv_id":"2005.03356","repositories_listed":1,"syntology":null},{"url":"/paper/lifeqa-a-real-life-dataset-for-video-question","slug":"lifeqa-a-real-life-dataset-for-video-question","title":"LifeQA: A Real-life Dataset for Video Question Answering","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/noise-estimation-using-density-estimation-for","slug":"noise-estimation-using-density-estimation-for","title":"Noise Estimation Using Density Estimation for Self-Supervised Multimodal Learning","date":"2020-03-06","arxiv_id":"2003.03186","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-conditional-relation-networks","slug":"hierarchical-conditional-relation-networks","title":"Hierarchical Conditional Relation Networks for Video Question Answering","date":"2020-02-25","arxiv_id":"2002.10698","repositories_listed":1,"syntology":null},{"url":"/paper/activitynet-qa-a-dataset-for-understanding","slug":"activitynet-qa-a-dataset-for-understanding","title":"ActivityNet-QA: A Dataset for Understanding Complex Web Videos via Question Answering","date":"2019-06-06","arxiv_id":"1906.02467","repositories_listed":1,"syntology":null},{"url":"/paper/heterogeneous-memory-enhanced-multimodal","slug":"heterogeneous-memory-enhanced-multimodal","title":"Heterogeneous Memory Enhanced Multimodal Attention Model for Video Question Answering","date":"2019-04-08","arxiv_id":"1904.04357","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/heterogeneous-memory-enhanced-multimodal#ran","syntology_url":"https://syntology.ai/paper/1904.04357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.04357"}},"official":{"repos":["fanchenyou/HME-VideoQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-choice-of-plausible-alternatives-an","slug":"visual-choice-of-plausible-alternatives-an","title":"Visual Choice of Plausible Alternatives: An Evaluation of Image-based Commonsense Causal Reasoning","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/memexqa-visual-memex-question-answering","slug":"memexqa-visual-memex-question-answering","title":"MemexQA: Visual Memex Question Answering","date":"2017-08-04","arxiv_id":"1708.01336","repositories_listed":1,"syntology":null},{"url":null,"slug":"how-far-can-off-the-shelf-multimodal-large","title":"How Far Can Off-the-Shelf Multimodal Large Language Models Go in Online Episodic Memory Question Answering?","date":"2025-06-19","arxiv_id":"2506.16450","repositories_listed":0,"syntology":null},{"url":null,"slug":"cogstream-context-guided-streaming-video","title":"CogStream: Context-guided Streaming Video Question Answering","date":"2025-06-12","arxiv_id":"2506.10516","repositories_listed":0,"syntology":null},{"url":null,"slug":"grid-logat-grid-based-local-and-global-area","title":"Grid-LOGAT: Grid Based Local and Global Area Transcription for Video Question Answering","date":"2025-05-30","arxiv_id":"2505.24371","repositories_listed":0,"syntology":null},{"url":null,"slug":"vudg-a-dataset-for-video-understanding-domain","title":"VUDG: A Dataset for Video Understanding Domain Generalization","date":"2025-05-30","arxiv_id":"2505.24346","repositories_listed":0,"syntology":null},{"url":null,"slug":"livevlm-efficient-online-video-understanding","title":"LiveVLM: Efficient Online Video Understanding via Streaming-Oriented KV Cache and Retrieval","date":"2025-05-21","arxiv_id":"2505.15269","repositories_listed":0,"syntology":null},{"url":null,"slug":"surveillancevqa-589k-a-benchmark-for","title":"SurveillanceVQA-589K: A Benchmark for Comprehensive Surveillance Video-Language Understanding with Large Models","date":"2025-05-19","arxiv_id":"2505.12589","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-complexity-in-videoqa-via","title":"Understanding Complexity in VideoQA via Visual Program Generation","date":"2025-05-19","arxiv_id":"2505.13429","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-the-nlpcc-2025-shared-task-4","title":"Overview of the NLPCC 2025 Shared Task 4: Multi-modal, Multilingual, and Multi-hop Medical Instructional Video Question Answering Challenge","date":"2025-05-11","arxiv_id":"2505.06814","repositories_listed":0,"syntology":null},{"url":"/paper/seed1-5-vl-technical-report","slug":"seed1-5-vl-technical-report","title":"Seed1.5-VL Technical Report","date":"2025-05-11","arxiv_id":"2505.07062","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-camera-motions-in-any","title":"Towards Understanding Camera Motions in Any Video","date":"2025-04-21","arxiv_id":"2504.15376","repositories_listed":0,"syntology":null},{"url":"/paper/self-alignment-of-large-video-language-models","slug":"self-alignment-of-large-video-language-models","title":"Self-alignment of Large Video Language Models with Refined Regularized Preference Optimization","date":"2025-04-16","arxiv_id":"2504.12083","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-can-objects-help-video-language","title":"How Can Objects Help Video-Language Understanding?","date":"2025-04-10","arxiv_id":"2504.07454","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-egocentric-video-question-answering","title":"Advancing Egocentric Video Question Answering with Multimodal Large Language Models","date":"2025-04-06","arxiv_id":"2504.04550","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-static-relationships-for-intra","title":"Leveraging Static Relationships for Intra-Type and Inter-Type Message Passing in Video Question Answering","date":"2025-04-03","arxiv_id":"2504.02417","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-llms-with-iterative-loop-structure","title":"Leveraging LLMs with Iterative Loop Structure for Enhanced Social Intelligence in Video Question Answering","date":"2025-03-27","arxiv_id":"2503.21190","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-res-self-reflection-in-large-vision","title":"Self-ReS: Self-Reflection in Large Vision-Language Models for Long Video Understanding","date":"2025-03-26","arxiv_id":"2503.20362","repositories_listed":0,"syntology":null},{"url":null,"slug":"logic-in-frames-dynamic-keyframe-search-via","title":"Logic-in-Frames: Dynamic Keyframe Search via Visual Semantic-Logical Verification for Long Video Understanding","date":"2025-03-17","arxiv_id":"2503.13139","repositories_listed":0,"syntology":null},{"url":null,"slug":"vited-video-temporal-evidence-distillation","title":"VITED: Video Temporal Evidence Distillation","date":"2025-03-17","arxiv_id":"2503.12855","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-temporal-sensitive-multi-dimensional","title":"TIME: Temporal-sensitive Multi-dimensional Instruction Tuning and Benchmarking for Video-LLMs","date":"2025-03-13","arxiv_id":"2503.09994","repositories_listed":0,"syntology":null},{"url":null,"slug":"everything-can-be-described-in-words-a-simple","title":"Everything Can Be Described in Words: A Simple Unified Multi-Modal Framework with Semantic and Temporal Alignment","date":"2025-03-12","arxiv_id":"2503.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-fine-grained-video-question-answering","title":"Towards Fine-Grained Video Question Answering","date":"2025-03-10","arxiv_id":"2503.06820","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-free-video-segmentation-for-vision","title":"Parameter-free Video Segmentation for Vision and Language Understanding","date":"2025-03-03","arxiv_id":"2503.01201","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-llm-based-video-frame-selection-for","title":"M-LLM Based Video Frame Selection for Efficient Video Understanding","date":"2025-02-27","arxiv_id":"2502.19680","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-retrieval-augmentation-for-open","title":"Multi-Modal Retrieval Augmentation for Open-Ended and Knowledge-Intensive Video Question Answering","date":"2025-02-17","arxiv_id":"2502.11747","repositories_listed":0,"syntology":null},{"url":"/paper/enter-event-based-interpretable-reasoning-for","slug":"enter-event-based-interpretable-reasoning-for","title":"ENTER: Event Based Interpretable Reasoning for VideoQA","date":"2025-01-24","arxiv_id":"2501.14194","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasvqa-advancing-videoqa-with-imperfect","title":"ReasVQA: Advancing VideoQA with Imperfect Reasoning Process","date":"2025-01-23","arxiv_id":"2501.13536","repositories_listed":0,"syntology":null},{"url":null,"slug":"admitting-ignorance-helps-the-video-question","title":"Admitting Ignorance Helps the Video Question Answering Models to Answer","date":"2025-01-15","arxiv_id":"2501.08771","repositories_listed":0,"syntology":null},{"url":null,"slug":"timelogic-a-temporal-logic-benchmark-for","title":"TimeLogic: A Temporal Logic Benchmark for Video QA","date":"2025-01-13","arxiv_id":"2501.07214","repositories_listed":0,"syntology":null},{"url":null,"slug":"commonsense-video-question-answering-through","title":"Commonsense Video Question Answering through Video-Grounded Entailment Tree Reasoning","date":"2025-01-09","arxiv_id":"2501.05069","repositories_listed":0,"syntology":null},{"url":null,"slug":"adacm-2-on-understanding-extremely-long-term-1","title":"AdaCM^2: On Understanding Extremely Long-Term Video with Adaptive Cross-Modality Memory Reduction","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-motion-aware-video-mllm","title":"Efficient Motion-Aware Video MLLM","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-video-llm-reasoning-via-agent-of","title":"Enhancing Video-LLM Reasoning via Agent-of-Thoughts Distillation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-frame-selection-for-efficient-video","title":"Flexible Frame Selection for Efficient Video Reasoning","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"perceive-query-reason-enhancing-video-qa-with","title":"Perceive, Query & Reason: Enhancing Video QA with Question-Guided Temporal Queries","date":"2024-12-26","arxiv_id":"2412.19304","repositories_listed":0,"syntology":null},{"url":null,"slug":"polysmart-trecvid-2024-medical-video-question","title":"PolySmart @ TRECVid 2024 Medical Video Question Answering","date":"2024-12-20","arxiv_id":"2412.15514","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-trec-2024-medical-video-question","title":"Overview of TREC 2024 Medical Video Question Answering (MedVidQA) Track","date":"2024-12-15","arxiv_id":"2412.11056","repositories_listed":0,"syntology":null},{"url":null,"slug":"iqvic-in-context-question-adaptive-vision","title":"IQViC: In-context, Question Adaptive Vision Compressor for Long-term Video Understanding LMMs","date":"2024-12-13","arxiv_id":"2412.09907","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundation-models-and-adaptive-feature","title":"Foundation Models and Adaptive Feature Selection: A Synergistic Approach to Video Question Answering","date":"2024-12-12","arxiv_id":"2412.09230","repositories_listed":0,"syntology":null},{"url":null,"slug":"seal-semantic-attention-learning-for-long","title":"SEAL: Semantic Attention Learning for Long Video Representation","date":"2024-12-02","arxiv_id":"2412.01798","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-video-llm-via-agent-of-thoughts","title":"Unlocking Video-LLM via Agent-of-Thoughts Distillation","date":"2024-12-02","arxiv_id":"2412.01694","repositories_listed":0,"syntology":null},{"url":null,"slug":"actions-and-objects-pathways-for-domain","title":"Actions and Objects Pathways for Domain Adaptation in Video Question Answering","date":"2024-11-29","arxiv_id":"2411.19434","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-test-2024-challenge-summary-and-a","title":"Perception Test 2024: Challenge Summary and a Novel Hour-Long VideoQA Benchmark","date":"2024-11-29","arxiv_id":"2411.19941","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperglm-hypergraph-for-video-scene-graph","title":"HyperGLM: HyperGraph for Video Scene Graph Generation and Anticipation","date":"2024-11-27","arxiv_id":"2411.18042","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoorion-tokenizing-object-dynamics-in","title":"VideoOrion: Tokenizing Object Dynamics in Videos","date":"2024-11-25","arxiv_id":"2411.16156","repositories_listed":0,"syntology":null},{"url":null,"slug":"adacm-2-on-understanding-extremely-long-term","title":"AdaCM$^2$: On Understanding Extremely Long-Term Video with Adaptive Cross-Modality Memory Reduction","date":"2024-11-19","arxiv_id":"2411.12593","repositories_listed":0,"syntology":null},{"url":null,"slug":"evqascore-efficient-video-question-answering","title":"EVQAScore: Efficient Video Question Answering Data Evaluation","date":"2024-11-11","arxiv_id":"2411.06908","repositories_listed":0,"syntology":null},{"url":null,"slug":"poze-sports-technique-feedback-under-data","title":"Poze: Sports Technique Feedback under Data Constraints","date":"2024-11-08","arxiv_id":"2411.05734","repositories_listed":0,"syntology":null},{"url":null,"slug":"flaash-flow-attention-adaptive-semantic","title":"FLAASH: Flow-Attention Adaptive Semantic Hierarchical Fusion for Multi-Modal Tobacco Content Analysis","date":"2024-10-25","arxiv_id":"2410.19896","repositories_listed":0,"syntology":null},{"url":"/paper/gpt-4o-system-card","slug":"gpt-4o-system-card","title":"GPT-4o System Card","date":"2024-10-25","arxiv_id":"2410.21276","repositories_listed":0,"syntology":null},{"url":null,"slug":"xgen-mm-vid-blip-3-video-you-only-need-32","title":"xGen-MM-Vid (BLIP-3-Video): You Only Need 32 Tokens to Represent a Video Even in VLMs","date":"2024-10-21","arxiv_id":"2410.16267","repositories_listed":0,"syntology":null}],"record_sha256":"77dadf568eeff40864db78b60f2629ce7ea202d98aab1a52d047e1cd80341411","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}