{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/8","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":22,"rows_per_page":100,"rows":[701,800],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/7","next":"/task/visual-question-answering/papers/9","papers":[{"url":"/paper/clevr-math-a-dataset-for-compositional","slug":"clevr-math-a-dataset-for-compositional","title":"CLEVR-Math: A Dataset for Compositional Language, Visual and Mathematical Reasoning","date":"2022-08-10","arxiv_id":"2208.05358","repositories_listed":1,"syntology":null},{"url":"/paper/chiqa-a-large-scale-image-based-real-world","slug":"chiqa-a-large-scale-image-based-real-world","title":"ChiQA: A Large Scale Image-based Real-World Question Answering Dataset for Multi-Modal Understanding","date":"2022-08-05","arxiv_id":"2208.03030","repositories_listed":1,"syntology":null},{"url":"/paper/prompt-tuning-for-generative-multimodal","slug":"prompt-tuning-for-generative-multimodal","title":"Prompt Tuning for Generative Multimodal Pretrained Models","date":"2022-08-04","arxiv_id":"2208.02532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompt-tuning-for-generative-multimodal#ran","syntology_url":"https://syntology.ai/paper/2208.02532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.02532"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tag-boosting-text-vqa-via-text-aware-visual","slug":"tag-boosting-text-vqa-via-text-aware-visual","title":"TAG: Boosting Text-VQA via Text-aware Visual Question-answer Generation","date":"2022-08-03","arxiv_id":"2208.01813","repositories_listed":1,"syntology":null},{"url":"/paper/generative-bias-for-visual-question-answering","slug":"generative-bias-for-visual-question-answering","title":"Generative Bias for Robust Visual Question Answering","date":"2022-08-01","arxiv_id":"2208.00690","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generative-bias-for-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2208.00690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.00690"}},"official":{"repos":["chojw/genb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lako-knowledge-driven-visual-question","slug":"lako-knowledge-driven-visual-question","title":"LaKo: Knowledge-driven Visual Question Answering via Late Knowledge-to-Text Injection","date":"2022-07-26","arxiv_id":"2207.12888","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lako-knowledge-driven-visual-question#ran","syntology_url":"https://syntology.ai/paper/2207.12888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12888"}},"official":{"repos":["hackerchenzhuo/LaKo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/winogavil-gamified-association-benchmark-to","slug":"winogavil-gamified-association-benchmark-to","title":"WinoGAViL: Gamified Association Benchmark to Challenge Vision-and-Language Models","date":"2022-07-25","arxiv_id":"2207.12576","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/winogavil-gamified-association-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2207.12576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12576"}},"official":{"repos":["winogavil/winogavil-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-data-augmentation-for-robust","slug":"rethinking-data-augmentation-for-robust","title":"Rethinking Data Augmentation for Robust Visual Question Answering","date":"2022-07-18","arxiv_id":"2207.08739","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/rethinking-data-augmentation-for-robust#ran","syntology_url":"https://syntology.ai/paper/2207.08739","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08739"}},"official":{"repos":["itemzheng/kddaug"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/clover-towards-a-unified-video-language","slug":"clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","arxiv_id":"2207.07885","repositories_listed":1,"syntology":null},{"url":"/paper/multiview-contrastive-learning-for-completely","slug":"multiview-contrastive-learning-for-completely","title":"Multiview Contrastive Learning for Completely Blind Video Quality Assessment of User Generated Content","date":"2022-07-13","arxiv_id":"2207.06148","repositories_listed":1,"syntology":null},{"url":"/paper/subjective-and-objective-quality-assessment-3","slug":"subjective-and-objective-quality-assessment-3","title":"Subjective and Objective Quality Assessment of High-Motion Sports Videos at Low-Bitrates","date":"2022-07-12","arxiv_id":"2207.05798","repositories_listed":1,"syntology":null},{"url":"/paper/video-graph-transformer-for-video-question","slug":"video-graph-transformer-for-video-question","title":"Video Graph Transformer for Video Question Answering","date":"2022-07-12","arxiv_id":"2207.05342","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":5,"n_ran_checked":8,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/video-graph-transformer-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2207.05342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05342"}},"official":{"repos":["sail-sg/vgt"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/viquae-a-dataset-for-knowledge-based-visual-1","slug":"viquae-a-dataset-for-knowledge-based-visual-1","title":"ViQuAE, a Dataset for Knowledge-based Visual Question Answering about Named Entities","date":"2022-07-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-effectiveness-of-video","slug":"exploring-the-effectiveness-of-video","title":"Exploring the Effectiveness of Video Perceptual Representation in Blind Video Quality Assessment","date":"2022-07-08","arxiv_id":"2207.03723","repositories_listed":1,"syntology":null},{"url":"/paper/knowing-earlier-what-right-means-to-you-a","slug":"knowing-earlier-what-right-means-to-you-a","title":"Knowing Earlier what Right Means to You: A Comprehensive VQA Dataset for Grounding Relative Directions via Multi-Task Learning","date":"2022-07-06","arxiv_id":"2207.02624","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-grounding-for-vqa-in-vision","slug":"weakly-supervised-grounding-for-vqa-in-vision","title":"Weakly Supervised Grounding for VQA in Vision-Language Transformers","date":"2022-07-05","arxiv_id":"2207.02334","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-end-to-end-retriever-reader","slug":"a-unified-end-to-end-retriever-reader","title":"A Unified End-to-End Retriever-Reader Framework for Knowledge-based VQA","date":"2022-06-30","arxiv_id":"2206.14989","repositories_listed":1,"syntology":null},{"url":"/paper/consistency-preserving-visual-question","slug":"consistency-preserving-visual-question","title":"Consistency-preserving Visual Question Answering in Medical Imaging","date":"2022-06-27","arxiv_id":"2206.13296","repositories_listed":1,"syntology":null},{"url":"/paper/visfis-visual-feature-importance-supervision","slug":"visfis-visual-feature-importance-supervision","title":"VisFIS: Visual Feature Importance Supervision with Right-for-the-Right-Reason Objectives","date":"2022-06-22","arxiv_id":"2206.11212","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visfis-visual-feature-importance-supervision#ran","syntology_url":"https://syntology.ai/paper/2206.11212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11212"}},"official":{"repos":["zfying/visfis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discovqa-temporal-distortion-content","slug":"discovqa-temporal-distortion-content","title":"DisCoVQA: Temporal Distortion-Content Transformers for Video Quality Assessment","date":"2022-06-20","arxiv_id":"2206.09853","repositories_listed":1,"syntology":null},{"url":"/paper/mixgen-a-new-multi-modal-data-augmentation","slug":"mixgen-a-new-multi-modal-data-augmentation","title":"MixGen: A New Multi-Modal Data Augmentation","date":"2022-06-16","arxiv_id":"2206.08358","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/mixgen-a-new-multi-modal-data-augmentation#ran","syntology_url":"https://syntology.ai/paper/2206.08358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08358"}},"official":{"repos":["amazon-research/mix-generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/language-models-are-general-purpose","slug":"language-models-are-general-purpose","title":"Language Models are General-Purpose Interfaces","date":"2022-06-13","arxiv_id":"2206.06336","repositories_listed":1,"syntology":null},{"url":"/paper/glipv2-unifying-localization-and-vision","slug":"glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","arxiv_id":"2206.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glipv2-unifying-localization-and-vision#ran","syntology_url":"https://syntology.ai/paper/2206.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05836"}},"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cvil-cross-lingual-training-of-vision","slug":"cvil-cross-lingual-training-of-vision","title":"cViL: Cross-Lingual Training of Vision-Language Models using Knowledge Distillation","date":"2022-06-07","arxiv_id":"2206.03354","repositories_listed":1,"syntology":null},{"url":"/paper/a-okvqa-a-benchmark-for-visual-question","slug":"a-okvqa-a-benchmark-for-visual-question","title":"A-OKVQA: A Benchmark for Visual Question Answering using World Knowledge","date":"2022-06-03","arxiv_id":"2206.01718","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-okvqa-a-benchmark-for-visual-question#ran","syntology_url":"https://syntology.ai/paper/2206.01718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01718"}},"official":{"repos":["allenai/aokvqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revive-regional-visual-representation-matters","slug":"revive-regional-visual-representation-matters","title":"REVIVE: Regional Visual Representation Matters in Knowledge-Based Visual Question Answering","date":"2022-06-02","arxiv_id":"2206.01201","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/revive-regional-visual-representation-matters#ran","syntology_url":"https://syntology.ai/paper/2206.01201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01201"}},"official":{"repos":["yzleroy/revive"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-efficient-modern-baseline-for-floodnet-vqa","slug":"an-efficient-modern-baseline-for-floodnet-vqa","title":"An Efficient Modern Baseline for FloodNet VQA","date":"2022-05-30","arxiv_id":"2205.15025","repositories_listed":1,"syntology":null},{"url":"/paper/git-a-generative-image-to-text-transformer","slug":"git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","arxiv_id":"2205.14100","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/git-a-generative-image-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.14100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14100"}},"official":{"repos":["microsoft/GenerativeImage2Text"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/pevl-position-enhanced-pre-training-and","slug":"pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","arxiv_id":"2205.11169","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pevl-position-enhanced-pre-training-and#ran","syntology_url":"https://syntology.ai/paper/2205.11169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11169"}},"official":{"repos":["thunlp/pevl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-neuro-symbolic-asp-pipeline-for-visual","slug":"a-neuro-symbolic-asp-pipeline-for-visual","title":"A Neuro-Symbolic ASP Pipeline for Visual Question Answering","date":"2022-05-16","arxiv_id":"2205.07548","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-answer-visual-questions-from-web","slug":"learning-to-answer-visual-questions-from-web","title":"Learning to Answer Visual Questions from Web Videos","date":"2022-05-10","arxiv_id":"2205.05019","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-to-answer-visual-questions-from-web#ran","syntology_url":"https://syntology.ai/paper/2205.05019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05019"}},"official":{"repos":["antoyang/just-ask"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/qlevr-a-diagnostic-dataset-for","slug":"qlevr-a-diagnostic-dataset-for","title":"QLEVR: A Diagnostic Dataset for Quantificational Language and Elementary Visual Reasoning","date":"2022-05-06","arxiv_id":"2205.03075","repositories_listed":1,"syntology":null},{"url":"/paper/declaration-based-prompt-tuning-for-visual","slug":"declaration-based-prompt-tuning-for-visual","title":"Declaration-based Prompt Tuning for Visual Question Answering","date":"2022-05-05","arxiv_id":"2205.02456","repositories_listed":1,"syntology":null},{"url":"/paper/laws-look-around-and-warm-start-natural","slug":"laws-look-around-and-warm-start-natural","title":"LAWS: Look Around and Warm-Start Natural Gradient Descent for Quantum Neural Networks","date":"2022-05-05","arxiv_id":"2205.02666","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-right-for-me-is-not-yet-right-for-you","slug":"what-is-right-for-me-is-not-yet-right-for-you","title":"What is Right for Me is Not Yet Right for You: A Dataset for Grounding Relative Directions via Multi-Task Learning","date":"2022-05-05","arxiv_id":"2205.02671","repositories_listed":1,"syntology":null},{"url":"/paper/textrm-dureader-textrm-vis-a-chinese-dataset","slug":"textrm-dureader-textrm-vis-a-chinese-dataset","title":"\\textrm{DuReader}_{\\textrm{vis}}: A Chinese Dataset for Open-domain Document Visual Question Answering","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vilmedic-a-framework-for-research-at-the","slug":"vilmedic-a-framework-for-research-at-the","title":"ViLMedic: a framework for research at the intersection of vision and language in medical AI","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/grit-general-robust-image-task-benchmark","slug":"grit-general-robust-image-task-benchmark","title":"GRIT: General Robust Image Task Benchmark","date":"2022-04-28","arxiv_id":"2204.13653","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grit-general-robust-image-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2204.13653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13653"}},"official":{"repos":["allenai/grit_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reliable-visual-question-answering-abstain","slug":"reliable-visual-question-answering-abstain","title":"Reliable Visual Question Answering: Abstain Rather Than Answer Incorrectly","date":"2022-04-28","arxiv_id":"2204.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reliable-visual-question-answering-abstain#ran","syntology_url":"https://syntology.ai/paper/2204.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13631"}},"official":{"repos":["facebookresearch/reliable_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/relvit-concept-guided-vision-transformer-for-1","slug":"relvit-concept-guided-vision-transformer-for-1","title":"RelViT: Concept-guided Vision Transformer for Visual Relational Reasoning","date":"2022-04-24","arxiv_id":"2204.11167","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/relvit-concept-guided-vision-transformer-for-1#ran","syntology_url":"https://syntology.ai/paper/2204.11167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11167"}},"official":{"repos":["NVlabs/RelViT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/hypergraph-transformer-weakly-supervised","slug":"hypergraph-transformer-weakly-supervised","title":"Hypergraph Transformer: Weakly-supervised Multi-hop Reasoning for Knowledge-based Visual Question Answering","date":"2022-04-22","arxiv_id":"2204.10448","repositories_listed":1,"syntology":null},{"url":"/paper/attention-in-reasoning-dataset-analysis-and","slug":"attention-in-reasoning-dataset-analysis-and","title":"Attention in Reasoning: Dataset, Analysis, and Modeling","date":"2022-04-20","arxiv_id":"2204.09774","repositories_listed":1,"syntology":null},{"url":"/paper/clevr-x-a-visual-reasoning-dataset-for","slug":"clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","date":"2022-04-05","arxiv_id":"2204.02380","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clevr-x-a-visual-reasoning-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2204.02380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02380"}},"official":{"repos":["explainableml/clevr-x"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/swapmix-diagnosing-and-regularizing-the-over","slug":"swapmix-diagnosing-and-regularizing-the-over","title":"SwapMix: Diagnosing and Regularizing the Over-Reliance on Visual Context in Visual Question Answering","date":"2022-04-05","arxiv_id":"2204.02285","repositories_listed":1,"syntology":null},{"url":"/paper/vl-interpret-an-interactive-visualization","slug":"vl-interpret-an-interactive-visualization","title":"VL-InterpreT: An Interactive Visualization Tool for Interpreting Vision-Language Transformers","date":"2022-03-30","arxiv_id":"2203.17247","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vl-interpret-an-interactive-visualization#ran","syntology_url":"https://syntology.ai/paper/2203.17247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.17247"}},"official":{"repos":["intellabs/vl-interpret"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/single-stream-multi-level-alignment-for","slug":"single-stream-multi-level-alignment-for","title":"Single-Stream Multi-Level Alignment for Vision-Language Pretraining","date":"2022-03-27","arxiv_id":"2203.14395","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-answer-questions-in-dynamic-audio","slug":"learning-to-answer-questions-in-dynamic-audio","title":"Learning to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2022-03-26","arxiv_id":"2203.14072","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/learning-to-answer-questions-in-dynamic-audio#ran","syntology_url":"https://syntology.ai/paper/2203.14072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.14072"}},"official":{"repos":["GeWu-Lab/MUSIC-AVQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/towards-efficient-and-elastic-visual-question","slug":"towards-efficient-and-elastic-visual-question","title":"Bilaterally Slimmable Transformer for Elastic and Efficient Visual Question Answering","date":"2022-03-24","arxiv_id":"2203.12814","repositories_listed":1,"syntology":null},{"url":"/paper/mukea-multimodal-knowledge-extraction-and","slug":"mukea-multimodal-knowledge-extraction-and","title":"MuKEA: Multimodal Knowledge Extraction and Accumulation for Knowledge-based Visual Question Answering","date":"2022-03-17","arxiv_id":"2203.09138","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mukea-multimodal-knowledge-extraction-and#ran","syntology_url":"https://syntology.ai/paper/2203.09138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09138"}},"official":{"repos":["andersonstra/mukea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/carets-a-consistency-and-robustness","slug":"carets-a-consistency-and-robustness","title":"CARETS: A Consistency And Robustness Evaluative Test Suite for VQA","date":"2022-03-15","arxiv_id":"2203.07613","repositories_listed":1,"syntology":null},{"url":"/paper/all-in-one-exploring-unified-video-language","slug":"all-in-one-exploring-unified-video-language","title":"All in One: Exploring Unified Video-Language Pre-training","date":"2022-03-14","arxiv_id":"2203.07303","repositories_listed":1,"syntology":null},{"url":"/paper/nlx-gpt-a-model-for-natural-language","slug":"nlx-gpt-a-model-for-natural-language","title":"NLX-GPT: A Model for Natural Language Explanations in Vision and Vision-Language Tasks","date":"2022-03-09","arxiv_id":"2203.05081","repositories_listed":1,"syntology":null},{"url":"/paper/barlow-constrained-optimization-for-visual","slug":"barlow-constrained-optimization-for-visual","title":"Barlow constrained optimization for Visual Question Answering","date":"2022-03-07","arxiv_id":"2203.03727","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-key-value-memory-enhanced-multi-step","slug":"dynamic-key-value-memory-enhanced-multi-step","title":"Dynamic Key-value Memory Enhanced Multi-step Graph Reasoning for Knowledge-based Visual Question Answering","date":"2022-03-06","arxiv_id":"2203.02985","repositories_listed":1,"syntology":null},{"url":"/paper/joint-answering-and-explanation-for-visual","slug":"joint-answering-and-explanation-for-visual","title":"Joint Answering and Explanation for Visual Commonsense Reasoning","date":"2022-02-25","arxiv_id":"2202.12626","repositories_listed":1,"syntology":null},{"url":"/paper/on-modality-bias-recognition-and-reduction","slug":"on-modality-bias-recognition-and-reduction","title":"On Modality Bias Recognition and Reduction","date":"2022-02-25","arxiv_id":"2202.12690","repositories_listed":1,"syntology":null},{"url":"/paper/og-sgg-ontology-guided-scene-graph-generation","slug":"og-sgg-ontology-guided-scene-graph-generation","title":"OG-SGG: Ontology-Guided Scene Graph Generation. A Case Study in Transfer Learning for Telepresence Robotics","date":"2022-02-21","arxiv_id":"2202.10201","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/delving-deeper-into-cross-lingual-visual","slug":"delving-deeper-into-cross-lingual-visual","title":"Delving Deeper into Cross-lingual Visual Question Answering","date":"2022-02-15","arxiv_id":"2202.07630","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-answers-for-visual-questions-asked","slug":"grounding-answers-for-visual-questions-asked","title":"Grounding Answers for Visual Questions Asked by Visually Impaired People","date":"2022-02-04","arxiv_id":"2202.01993","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-answers-for-visual-questions-asked#ran","syntology_url":"https://syntology.ai/paper/2202.01993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01993"}},"official":{"repos":["ccychongyanchen/vizwizvqagroundingcrowdsourcing"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compositionality-as-lexical-symmetry","slug":"compositionality-as-lexical-symmetry","title":"Compositionality as Lexical Symmetry","date":"2022-01-30","arxiv_id":"2201.12926","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/compositionality-as-lexical-symmetry#ran","syntology_url":"https://syntology.ai/paper/2201.12926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12926"}},"official":{"repos":["ekinakyurek/lexsym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-module-networks-for-systematic","slug":"transformer-module-networks-for-systematic","title":"Transformer Module Networks for Systematic Generalization in Visual Question Answering","date":"2022-01-27","arxiv_id":"2201.11316","repositories_listed":1,"syntology":null},{"url":"/paper/faver-blind-quality-prediction-of-variable","slug":"faver-blind-quality-prediction-of-variable","title":"FAVER: Blind Quality Prediction of Variable Frame Rate Videos","date":"2022-01-05","arxiv_id":"2201.01492","repositories_listed":1,"syntology":null},{"url":"/paper/maintaining-reasoning-consistency-in","slug":"maintaining-reasoning-consistency-in","title":"Maintaining Reasoning Consistency in Compositional Visual Question Answering","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/query-and-attention-augmentation-for","slug":"query-and-attention-augmentation-for","title":"Query and Attention Augmentation for Knowledge-Based Explainable Reasoning","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-image-visual-question-answering","slug":"multi-image-visual-question-answering","title":"Multi-Image Visual Question Answering","date":"2021-12-27","arxiv_id":"2112.13706","repositories_listed":1,"syntology":null},{"url":"/paper/latr-layout-aware-transformer-for-scene-text","slug":"latr-layout-aware-transformer-for-scene-text","title":"LaTr: Layout-Aware Transformer for Scene-Text VQA","date":"2021-12-23","arxiv_id":"2112.12494","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/latr-layout-aware-transformer-for-scene-text#ran","syntology_url":"https://syntology.ai/paper/2112.12494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12494"}},"official":null}},{"url":"/paper/clevr3d-compositional-language-and-elementary","slug":"clevr3d-compositional-language-and-elementary","title":"Comprehensive Visual Question Answering on Point Clouds through Compositional Scene Manipulation","date":"2021-12-22","arxiv_id":"2112.11691","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clevr3d-compositional-language-and-elementary#ran","syntology_url":"https://syntology.ai/paper/2112.11691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.11691"}},"official":{"repos":["yanx27/clevr3d"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/general-greedy-de-bias-learning","slug":"general-greedy-de-bias-learning","title":"General Greedy De-bias Learning","date":"2021-12-20","arxiv_id":"2112.10572","repositories_listed":1,"syntology":null},{"url":"/paper/scanqa-3d-question-answering-for-spatial","slug":"scanqa-3d-question-answering-for-spatial","title":"ScanQA: 3D Question Answering for Spatial Scene Understanding","date":"2021-12-20","arxiv_id":"2112.10482","repositories_listed":1,"syntology":null},{"url":"/paper/align-and-prompt-video-and-language-pre","slug":"align-and-prompt-video-and-language-pre","title":"Align and Prompt: Video-and-Language Pre-training with Entity Prompts","date":"2021-12-17","arxiv_id":"2112.09583","repositories_listed":1,"syntology":null},{"url":"/paper/kat-a-knowledge-augmented-transformer-for","slug":"kat-a-knowledge-augmented-transformer-for","title":"KAT: A Knowledge Augmented Transformer for Vision-and-Language","date":"2021-12-16","arxiv_id":"2112.08614","repositories_listed":1,"syntology":null},{"url":"/paper/bilateral-cross-modality-graph-matching","slug":"bilateral-cross-modality-graph-matching","title":"Bilateral Cross-Modality Graph Matching Attention for Feature Fusion in Visual Question Answering","date":"2021-12-14","arxiv_id":"2112.07270","repositories_listed":1,"syntology":null},{"url":"/paper/dual-key-multimodal-backdoors-for-visual","slug":"dual-key-multimodal-backdoors-for-visual","title":"Dual-Key Multimodal Backdoors for Visual Question Answering","date":"2021-12-14","arxiv_id":"2112.07668","repositories_listed":1,"syntology":null},{"url":"/paper/change-detection-meets-visual-question","slug":"change-detection-meets-visual-question","title":"Change Detection Meets Visual Question Answering","date":"2021-12-12","arxiv_id":"2112.06343","repositories_listed":1,"syntology":null},{"url":"/paper/video-as-conditional-graph-hierarchy-for","slug":"video-as-conditional-graph-hierarchy-for","title":"Video as Conditional Graph Hierarchy for Multi-Granular Question Answering","date":"2021-12-12","arxiv_id":"2112.06197","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-as-conditional-graph-hierarchy-for#ran","syntology_url":"https://syntology.ai/paper/2112.06197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06197"}},"official":{"repos":["doc-doc/hqga"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mlp-architectures-for-vision-and-language","slug":"mlp-architectures-for-vision-and-language","title":"MLP Architectures for Vision-and-Language Modeling: An Empirical Study","date":"2021-12-08","arxiv_id":"2112.04453","repositories_listed":1,"syntology":null},{"url":"/paper/debiased-visual-question-answering-from","slug":"debiased-visual-question-answering-from","title":"Debiased Visual Question Answering from Feature and Sample Perspectives","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/classification-regression-for-chart","slug":"classification-regression-for-chart","title":"Classification-Regression for Chart Comprehension","date":"2021-11-29","arxiv_id":"2111.14792","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/classification-regression-for-chart#ran","syntology_url":"https://syntology.ai/paper/2111.14792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14792"}},"official":{"repos":["levymsn/cqa-crct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/many-heads-but-one-brain-an-overview-of","slug":"many-heads-but-one-brain-an-overview-of","title":"Many Heads but One Brain: Fusion Brain -- a Competition and a Single Multimodal Multitask Architecture","date":"2021-11-22","arxiv_id":"2111.10974","repositories_listed":1,"syntology":null},{"url":"/paper/blind-vqa-on-360deg-video-via-progressively","slug":"blind-vqa-on-360deg-video-via-progressively","title":"Blind VQA on 360° Video via Progressively Learning from Pixels, Frames and Video","date":"2021-11-18","arxiv_id":"2111.09503","repositories_listed":1,"syntology":null},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/no-reference-video-quality-assessment-based","slug":"no-reference-video-quality-assessment-based","title":"No-Reference Video Quality Assessment Based on Benford’s Law and Perceptual Features","date":"2021-11-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/introspective-distillation-for-robust","slug":"introspective-distillation-for-robust","title":"Introspective Distillation for Robust Question Answering","date":"2021-11-01","arxiv_id":"2111.01026","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/introspective-distillation-for-robust#ran","syntology_url":"https://syntology.ai/paper/2111.01026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.01026"}},"official":{"repos":["yuleiniu/introd"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mirtt-learning-multimodal-interaction","slug":"mirtt-learning-multimodal-interaction","title":"MIRTT: Learning Multimodal Interaction Representations from Trilinear Transformers for Visual Question Answering","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/alignment-attention-by-matching-key-and-query","slug":"alignment-attention-by-matching-key-and-query","title":"Alignment Attention by Matching Key and Query Distributions","date":"2021-10-25","arxiv_id":"2110.12567","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/alignment-attention-by-matching-key-and-query#ran","syntology_url":"https://syntology.ai/paper/2110.12567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.12567"}},"official":{"repos":["szhang42/alignment_attention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","slug":"iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","arxiv_id":"2110.13214","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/iconqa-a-new-benchmark-for-abstract-diagram#ran","syntology_url":"https://syntology.ai/paper/2110.13214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13214"}},"official":{"repos":["lupantech/iconqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dair-data-augmented-invariant-regularization-1","slug":"dair-data-augmented-invariant-regularization-1","title":"Robustness through Data Augmentation Loss Consistency","date":"2021-10-21","arxiv_id":"2110.11205","repositories_listed":1,"syntology":null},{"url":"/paper/towards-language-guided-visual-recognition","slug":"towards-language-guided-visual-recognition","title":"Towards Language-guided Visual Recognition via Dynamic Convolutions","date":"2021-10-17","arxiv_id":"2110.08797","repositories_listed":1,"syntology":null},{"url":"/paper/a-good-prompt-is-worth-millions-of-parameters","slug":"a-good-prompt-is-worth-millions-of-parameters","title":"A Good Prompt Is Worth Millions of Parameters: Low-resource Prompt-based Learning for Vision-Language Models","date":"2021-10-16","arxiv_id":"2110.08484","repositories_listed":1,"syntology":null},{"url":"/paper/semantically-distributed-robust-optimization","slug":"semantically-distributed-robust-optimization","title":"Semantically Distributed Robust Optimization for Vision-and-Language Inference","date":"2021-10-14","arxiv_id":"2110.07165","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-accuracy-a-consolidated-tool-for","slug":"beyond-accuracy-a-consolidated-tool-for","title":"Beyond Accuracy: A Consolidated Tool for Visual Question Answering Benchmarking","date":"2021-10-11","arxiv_id":"2110.05159","repositories_listed":1,"syntology":null},{"url":"/paper/pano-avqa-grounded-audio-visual-question-1","slug":"pano-avqa-grounded-audio-visual-question-1","title":"Pano-AVQA: Grounded Audio-Visual Question Answering on 360$^\\circ$ Videos","date":"2021-10-11","arxiv_id":"2110.05122","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-samples-synthesizing-and","slug":"counterfactual-samples-synthesizing-and","title":"Counterfactual Samples Synthesizing and Training for Robust Visual Question Answering","date":"2021-10-03","arxiv_id":"2110.01013","repositories_listed":1,"syntology":null},{"url":"/paper/proto-program-guided-transformer-for-program","slug":"proto-program-guided-transformer-for-program","title":"ProTo: Program-Guided Transformer for Program-Guided Tasks","date":"2021-10-02","arxiv_id":"2110.00804","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/proto-program-guided-transformer-for-program#ran","syntology_url":"https://syntology.ai/paper/2110.00804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.00804"}},"official":{"repos":["sjtuytc/Neurips21-ProTo-Program-guided-Transformers-for-Program-guided-Tasks"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/calibrating-concepts-and-operations-towards","slug":"calibrating-concepts-and-operations-towards","title":"Calibrating Concepts and Operations: Towards Symbolic Reasoning on Real Images","date":"2021-10-01","arxiv_id":"2110.00519","repositories_listed":1,"syntology":null},{"url":"/paper/the-spoon-is-in-the-sink-assisting-visually","slug":"the-spoon-is-in-the-sink-assisting-visually","title":"The Spoon Is in the Sink: Assisting Visually Impaired People in the Kitchen","date":"2021-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/does-vision-and-language-pretraining-improve","slug":"does-vision-and-language-pretraining-improve","title":"Does Vision-and-Language Pretraining Improve Lexical Grounding?","date":"2021-09-21","arxiv_id":"2109.10246","repositories_listed":1,"syntology":null}],"record_sha256":"4c941eaeb7868bdbba3baa8e21cb943b448d363b61797447b79db24e768af53c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}