{"url":"/dataset/gqa","name":"GQA","full_name":"GQA","description_markdown":"The **GQA** dataset is a large-scale visual question answering dataset with real images from the Visual Genome dataset and balanced question-answer pairs. Each training and validation image is also associated with scene graph annotations describing the classes and attributes of those objects in the scene, and their pairwise relations. Along with the images and question-answer pairs, the GQA dataset provides two types of pre-extracted visual features for each image – convolutional grid features of size 7×7×2048 extracted from a ResNet-101 network trained on ImageNet, and object detection features of size Ndet×2048 (where Ndet is the number of detected objects in each image with a maximum of 100 per image) from a Faster R-CNN detector.\r\n\r\nSource: [Language-Conditioned Graph Networks for Relational Reasoning](https://arxiv.org/abs/1905.04405)\r\nImage Source: [https://arxiv.org/pdf/1902.09506.pdf](https://arxiv.org/pdf/1902.09506.pdf)","description_withheld":null,"homepage":"https://cs.stanford.edu/people/dorarad/gqa/","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/gqa-a-new-dataset-for-compositional-question","title":"GQA: A New Dataset for Real-World Visual Reasoning and Compositional Question Answering","first_author":"Drew A. Hudson","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Visual Question Answering","url":"/task/visual-question-answering-1","datasets_with_task":"/datasets/task/visual-question-answering-1"},{"name":"Scene Graph Generation","url":"/task/scene-graph-generation","datasets_with_task":"/datasets/task/scene-graph-generation"},{"name":"Graph Question Answering","url":"/task/graph-question-answering","datasets_with_task":"/datasets/task/graph-question-answering"}],"languages":[],"variants":["GQA test-std","GQA test-dev","GQA Test2019","GQA","GQA-OOD"],"data_loaders":[{"repo":"https://github.com/allenai/allennlp-models","url":"https://docs.allennlp.org/models/main/models/vision/dataset_readers/gqa/","frameworks":["pytorch"]}],"num_papers_in_archive":749,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-gqa-test2019","task":"Visual Question Answering (VQA)","dataset_variant":"GQA Test2019","rows":127,"metrics":["Accuracy","Binary","Open","Consistency","Plausibility","Validity","Distribution"],"first_row_in_archive_order":{"model":"human","paper":null,"metrics":{"Accuracy":"89.3","Binary":"91.2","Consistency":"98.4","Distribution":"0.0","Open":"87.4","Plausibility":"97.2","Validity":"98.9"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-gqa-test-dev","task":"Visual Question Answering (VQA)","dataset_variant":"GQA test-dev","rows":17,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CFR","paper":"/paper/coarse-to-fine-reasoning-for-visual-question","metrics":{"Accuracy":"72.1"},"code_links":[{"title":"aioz-ai/cfr_vqa","url":"https://github.com/aioz-ai/cfr_vqa"},{"title":"aioz-ai/crf_vqa","url":"https://github.com/aioz-ai/crf_vqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-gqa-test-std","task":"Visual Question Answering (VQA)","dataset_variant":"GQA test-std","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ProTo","paper":"/paper/proto-program-guided-transformer-for-program","metrics":{"Accuracy":"65.14"},"code_links":[{"title":"sjtuytc/Neurips21-ProTo-Program-guided-Transformers-for-Program-guided-Tasks","url":"https://github.com/sjtuytc/Neurips21-ProTo-Program-guided-Transformers-for-Program-guided-Tasks"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/graph-question-answering-on-gqa","task":"Graph Question Answering","dataset_variant":"GQA","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GraphVQA","paper":"/paper/graghvqa-language-guided-graph-neural","metrics":{"Accuracy":"96.30"},"code_links":[{"title":"codexxxl/GraphVQA","url":"https://github.com/codexxxl/GraphVQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-gqa","task":"Visual Question Answering (VQA)","dataset_variant":"GQA","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PEVL+","paper":"/paper/pevl-position-enhanced-pre-training-and","metrics":{"Accuracy":"77"},"code_links":[{"title":"thunlp/pevl","url":"https://github.com/thunlp/pevl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-gqa","task":"Object Detection","dataset_variant":"GQA","rows":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"KnowZRel","paper":"/paper/knowzrel-common-sense-knowledge-based-zero","metrics":{"mAP":"39"},"code_links":[{"title":"jaleedkhan/zsrr-sgg","url":"https://github.com/jaleedkhan/zsrr-sgg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-graph-generation-on-gqa","task":"Scene Graph Generation","dataset_variant":"GQA","rows":1,"metrics":["zR@100","zR@20","zR@50"],"first_row_in_archive_order":{"model":"KnowZRel","paper":"/paper/knowzrel-common-sense-knowledge-based-zero","metrics":{"zR@100":"29.56","zR@20":"12.47","zR@50":"22.51"},"code_links":[{"title":"jaleedkhan/zsrr-sgg","url":"https://github.com/jaleedkhan/zsrr-sgg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-gqa-1","task":"Visual Question Answering","dataset_variant":"GQA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"LocVLM-L","paper":"/paper/learning-to-localize-objects-improves-spatial","metrics":{"Accuracy":"50.2"},"code_links":[{"title":"kahnchana/locvlm","url":"https://github.com/kahnchana/locvlm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/knowzrel-common-sense-knowledge-based-zero","title":"KnowZRel: Common Sense Knowledge-based Zero-Shot Relationship Retrieval for Generalised Scene Graph Generation","date":"2025-02-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-localize-objects-improves-spatial","title":"Learning to Localize Objects Improves Spatial Reasoning in Visual-LLMs","date":"2024-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lyrics-boosting-fine-grained-language-vision","title":"Lyrics: Boosting Fine-grained Language-Vision Alignment and Comprehension via Semantic-aware Visual Objects","date":"2023-12-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/visual-program-distillation-distilling-tools","title":"Visual Program Distillation: Distilling Tools and Programmatic Reasoning into Vision-Language Models","date":"2023-12-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vinvl-l-enriching-visual-representation-with","title":"VinVL+L: Enriching Visual Representation with Location Context in VQA","date":"2023-02-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","rows_on_this_dataset":6,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining","title":"Plug-and-Play VQA: Zero-shot VQA by Conjoining Large Pretrained Models with Zero Training","date":"2022-10-17","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/relvit-concept-guided-vision-transformer-for-1","title":"RelViT: Concept-guided Vision Transformer for Visual Relational Reasoning","date":"2022-04-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":6,"samples_unverified":4,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-good-prompt-is-worth-millions-of-parameters","title":"A Good Prompt Is Worth Millions of Parameters: Low-resource Prompt-based Learning for Vision-Language Models","date":"2021-10-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/coarse-to-fine-reasoning-for-visual-question","title":"Coarse-to-Fine Reasoning for Visual Question Answering","date":"2021-10-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/proto-program-guided-transformer-for-program","title":"ProTo: Program-Guided Transformer for Program-Guided Tasks","date":"2021-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/graghvqa-language-guided-graph-neural","title":"GraghVQA: Language-Guided Graph Neural Networks for Graph-based Visual Question Answering","date":"2021-04-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vinvl-making-visual-representations-matter-in","title":"VinVL: Revisiting Visual Representations in Vision-Language Models","date":"2021-01-02","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lxmert-learning-cross-modality-encoder","title":"LXMERT: Learning Cross-Modality Encoder Representations from Transformers","date":"2019-08-20","rows_on_this_dataset":4,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":4,"samples_unverified":11,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/graph-reasoning-networks-for-visual-question","title":"Bilinear Graph Networks for Visual Question Answering","date":"2019-07-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-by-abstraction-the-neural-state","title":"Learning by Abstraction: The Neural State Machine","date":"2019-07-09","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":3,"samples_unverified":18,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-conditioned-graph-networks-for","title":"Language-Conditioned Graph Networks for Relational Reasoning","date":"2019-05-10","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gqa-a-new-dataset-for-compositional-question","title":"GQA: A New Dataset for Real-World Visual Reasoning and Compositional Question Answering","date":"2019-02-25","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/bottom-up-and-top-down-attention-for-image","title":"Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering","date":"2017-07-25","rows_on_this_dataset":1,"code_links":65,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":9,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":15,"samples_harvested":131,"samples_ran":61,"samples_unverified":70,"pointer_only_for_licence":31,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}