{"url":"/dataset/scanqa","name":"ScanQA","full_name":"ScanQA: 3D Question Answering for Spatial Scene Understanding","description_markdown":"We collected 41,363 questions and 58,191 answers, in-\r\ncluding 32,337 unique questions and 16,999 unique an-\r\nswers. Table 2 presents the statistics of the ScanQA\r\ndataset. This dataset is an order of magnitude larger than\r\nexisting embodied question-answering datasets in terms of\r\nboth question size and variation. For example, the EQA\r\ndataset contains 4,246 questions, consisting of 147\r\nunique questions in its training set. The EQA-MP3D\r\ndataset contains 767 questions consisting of 174 unique\r\nquestions in its training set. Considering that our dataset\r\ncontains not only question–answer pairs but also 3D object\r\nlocalization annotations, we assume that this is the largest\r\ndataset to specify the nature of objects in 3D scenes with the\r\nquestion answering form. The distribution of the questions\r\nbased on their first word. We collected\r\nvarious types of questions through question auto-generation\r\nand editing by humans.","description_withheld":null,"homepage":"","introduced_date":"2021-12-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/scanqa-3d-question-answering-for-spatial","title":"ScanQA: 3D Question Answering for Spatial Scene Understanding","first_author":"Daichi Azuma","url":null},"license":null,"modalities":[],"tasks":[{"name":"3D Question Answering (3D-QA)","url":"/task/3d-question-answering-3d-qa","datasets_with_task":"/datasets/task/3d-question-answering-3d-qa"}],"languages":[],"variants":["ScanQA Test w/ objects","ScanQA"],"data_loaders":[],"num_papers_in_archive":70,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/3d-question-answering-3d-qa-on-scanqa-test-w","task":"3D Question Answering (3D-QA)","dataset_variant":"ScanQA Test w/ objects","rows":18,"metrics":["Exact Match","BLEU-1","BLEU-4","ROUGE","METEOR","CIDEr"],"first_row_in_archive_order":{"model":"BridgeQA","paper":"/paper/bridging-the-gap-between-2d-and-3d-visual","metrics":{"BLEU-1":"34.49","BLEU-4":"24.06","CIDEr":"83.75","Exact Match":"31.29","METEOR":"16.51","ROUGE":"43.26"},"code_links":[{"title":"matthewdm0816/bridgeqa","url":"https://github.com/matthewdm0816/bridgeqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/video-3d-llm-learning-position-aware-video","title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","date":"2024-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/llava-3d-a-simple-yet-effective-pathway-to","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","date":"2024-09-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/llava-onevision-easy-visual-task-transfer","title":"LLaVA-OneVision: Easy Visual Task Transfer","date":"2024-08-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/scene-llm-extending-language-model-for-3d","title":"Scene-LLM: Extending Language Model for 3D Visual Understanding and Reasoning","date":"2024-03-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":10,"samples_unverified":3,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chat-3d-v2-bridging-3d-scene-and-large","title":"Chat-Scene: Bridging 3D Scene and Large Language Models with Object Identifiers","date":"2023-12-13","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-learning-a-generalist-model-for","title":"Towards Learning a Generalist Model for Embodied Navigation","date":"2023-12-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-embodied-generalist-agent-in-3d-world","title":"An Embodied Generalist Agent in 3D World","date":"2023-11-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":3,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/3d-llm-injecting-the-3d-world-into-large","title":"3D-LLM: Injecting the 3D World into Large Language Models","date":"2023-07-24","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":5,"samples_unverified":5,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":51,"samples_ran":16,"samples_unverified":35,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scanqa-3d-question-answering-for-spatial","title":"ScanQA: 3D Question Answering for Spatial Scene Understanding","date":"2021-12-20","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":129,"samples_ran":66,"samples_unverified":63,"pointer_only_for_licence":28,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}