{"url":"/dataset/sqa3d","name":"SQA3D","full_name":"Situated Question Answering in 3D Scenes","description_markdown":"SQA3D is a dataset for embodied scene understanding, where an agent needs to comprehend the scene it situates  from an first person's perspective and answer questions. The questions are designed to be situated, embodied and knowledge-intensive. We offer three different modalities to represent a 3D scene: 3D scan, egocentric video and BEV picture.","description_withheld":null,"homepage":"https://sqa3d.github.io","introduced_date":"2023-01-30","introduced_date_note":null,"introduced_by":{"paper":"/paper/sqa3d-situated-question-answering-in-3d","title":"SQA3D: Situated Question Answering in 3D Scenes","first_author":"Xiaojian Ma","url":null},"license":{"name":"CC-BY-4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"3D","url":"/datasets/modality/3d"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"3D Question Answering (3D-QA)","url":"/task/3d-question-answering-3d-qa","datasets_with_task":"/datasets/task/3d-question-answering-3d-qa"},{"name":"Referring Expression","url":"/task/referring-expression","datasets_with_task":"/datasets/task/referring-expression"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SQA3D"],"data_loaders":[{"repo":"https://github.com/SilongYong/SQA3D","url":"https://github.com/SilongYong/SQA3D","frameworks":["pytorch"]}],"num_papers_in_archive":58,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/3d-question-answering-3d-qa-on-sqa3d","task":"3D Question Answering (3D-QA)","dataset_variant":"SQA3D","rows":13,"metrics":["Exact Match"],"first_row_in_archive_order":{"model":"LLaVA-3D","paper":"/paper/llava-3d-a-simple-yet-effective-pathway-to","metrics":{"Exact Match":"60.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-answering-on-sqa3d","task":"Question Answering","dataset_variant":"SQA3D","rows":7,"metrics":["AnswerExactMatch (Question Answering)"],"first_row_in_archive_order":{"model":"CREMA","paper":"/paper/crema-multimodal-compositional-video","metrics":{"AnswerExactMatch (Question Answering)":"54.6"},"code_links":[{"title":"Yui010206/CREMA","url":"https://github.com/Yui010206/CREMA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/referring-expression-on-sqa3d-1","task":"Referring Expression","dataset_variant":"SQA3D","rows":1,"metrics":["Acc@0.5m","Acc@1.0m","Acc@15°","Acc@30°"],"first_row_in_archive_order":{"model":"Random","paper":"/paper/sqa3d-situated-question-answering-in-3d","metrics":{"Acc@0.5m":"14.60","Acc@1.0m":"34.21","Acc@15°":"22.39","Acc@30°":"42.28"},"code_links":[{"title":"SilongYong/SQA3D","url":"https://github.com/SilongYong/SQA3D"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/video-3d-llm-learning-position-aware-video","title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","date":"2024-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-instruction-tuning-with-synthetic-data","title":"Video Instruction Tuning With Synthetic Data","date":"2024-10-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/llava-3d-a-simple-yet-effective-pathway-to","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","date":"2024-09-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lexicon3d-probing-visual-foundation-models","title":"Lexicon3D: Probing Visual Foundation Models for Complex 3D Scene Understanding","date":"2024-09-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llava-onevision-easy-visual-task-transfer","title":"LLaVA-OneVision: Easy Visual Task Transfer","date":"2024-08-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/situational-awareness-matters-in-3d-vision","title":"Situational Awareness Matters in 3D Vision Language Reasoning","date":"2024-06-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":10,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifying-3d-vision-language-understanding-via","title":"Unifying 3D Vision-Language Understanding via Promptable Queries","date":"2024-05-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scene-llm-extending-language-model-for-3d","title":"Scene-LLM: Extending Language Model for 3D Visual Understanding and Reasoning","date":"2024-03-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/crema-multimodal-compositional-video","title":"CREMA: Generalizable and Efficient Video-Language Reasoning via Multimodal Modular Fusion","date":"2024-02-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chat-3d-v2-bridging-3d-scene-and-large","title":"Chat-Scene: Bridging 3D Scene and Large Language Models with Object Identifiers","date":"2023-12-13","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-embodied-generalist-agent-in-3d-world","title":"An Embodied Generalist Agent in 3D World","date":"2023-11-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":3,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/frozen-transformers-in-language-models-are","title":"Frozen Transformers in Language Models Are Effective Visual Encoder Layers","date":"2023-10-19","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":8,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sqa3d-situated-question-answering-in-3d","title":"SQA3D: Situated Question Answering in 3D Scenes","date":"2022-10-14","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scanqa-3d-question-answering-for-spatial","title":"ScanQA: 3D Question Answering for Spatial Scene Understanding","date":"2021-12-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scan2cap-context-aware-dense-captioning-in","title":"Scan2Cap: Context-aware Dense Captioning in RGB-D Scans","date":"2020-12-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deep-modular-co-attention-networks-for-visual-1","title":"Deep Modular Co-Attention Networks for Visual Question Answering","date":"2019-06-25","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":94,"samples_ran":59,"samples_unverified":35,"pointer_only_for_licence":5,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}