{"url":"/task/3d-question-answering-3d-qa","name":"3D Question Answering (3D-QA)","slug":"3d-question-answering-3d-qa","description_markdown":"A 3D-QA task requires models to answer a question when given all the information of a 3D scene. Here, models use the 3D spatial information, such as RGB-D scans or point cloud data. We also require models to specify the 3D-bounding boxes of objects that are related to this question answering. This prevents models from answering questions by relying on the textual priors of the trained questions without examining the scene. However, unlike 3D dense captioning, we do not require models to target one described object for each question. This is because multiple objects can be used to answer certain questions. For example, the question “What color is the chairs around the table?” is related to multiple objects. This question is also answerable as long as the chairs around the unique table in the scene have the same color. In such scenarios, we require models to answer the question addressing multiple 3D-bounding boxes.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":22,"papers_with_code":17,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/3d-question-answering-3d-qa-on-scanqa-test-w","slug":"3d-question-answering-3d-qa-on-scanqa-test-w","dataset":"ScanQA Test w/ objects","dataset_url":"/dataset/scanqa","rows_in_archive":18,"metrics":["Exact Match","BLEU-1","BLEU-4","ROUGE","METEOR","CIDEr"],"first_row_in_archive_order":{"model":"BridgeQA","paper_title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","paper_url":"/paper/bridging-the-gap-between-2d-and-3d-visual","paper_date":"2024-02-24","arxiv_id":"2402.15933","code_links":[{"title":"matthewdm0816/bridgeqa","url":"https://github.com/matthewdm0816/bridgeqa"}],"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":13}}},{"leaderboard":"/sota/3d-question-answering-3d-qa-on-sqa3d","slug":"3d-question-answering-3d-qa-on-sqa3d","dataset":"SQA3D","dataset_url":"/dataset/sqa3d","rows_in_archive":13,"metrics":["Exact Match"],"first_row_in_archive_order":{"model":"LLaVA-3D","paper_title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","paper_url":"/paper/llava-3d-a-simple-yet-effective-pathway-to","paper_date":"2024-09-26","arxiv_id":"2409.18125","code_links":[],"syntology":null}},{"leaderboard":"/sota/3d-question-answering-3d-qa-on-3d-mm-vet","slug":"3d-question-answering-3d-qa-on-3d-mm-vet","dataset":"3D MM-Vet","dataset_url":"/dataset/3d-mm-vet","rows_in_archive":5,"metrics":["Overall Accuracy"],"first_row_in_archive_order":{"model":"ShapeLLM-13B","paper_title":"ShapeLLM: Universal 3D Object Understanding for Embodied Interaction","paper_url":"/paper/shapellm-universal-3d-object-understanding","paper_date":"2024-02-27","arxiv_id":"2402.17766","code_links":[{"title":"qizekun/ShapeLLM","url":"https://github.com/qizekun/ShapeLLM"},{"title":"qizekun/ReCon","url":"https://github.com/qizekun/ReCon"},{"title":"runpeidong/act","url":"https://github.com/runpeidong/act"}],"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/scanqa","name":"ScanQA","full_name":"ScanQA: 3D Question Answering for Spatial Scene Understanding","num_papers_in_archive":70},{"url":"/dataset/sqa3d","name":"SQA3D","full_name":"Situated Question Answering in 3D Scenes","num_papers_in_archive":58},{"url":"/dataset/3d-mm-vet","name":"3D MM-Vet","full_name":"","num_papers_in_archive":4},{"url":"/dataset/beacon3d","name":"Beacon3D","full_name":"","num_papers_in_archive":2},{"url":"/dataset/msqa","name":"MSQA","full_name":"","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[{"url":"/task/visual-question-answering","name":"Visual Question Answering (VQA)"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":17,"of":17,"tagged_in_all":22,"items":[{"url":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_unverified":35,"n_pointer_only":0}},{"url":"/paper/point-bind-point-llm-aligning-point-cloud","title":"Point-Bind & Point-LLM: Aligning Point Cloud with Multi-modality for 3D Understanding, Generation, and Instruction Following","date":"2023-09-01","arxiv_id":"2309.00615","repositories_listed":5,"syntology":{"n":20,"n_ran":13,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/3d-llm-injecting-the-3d-world-into-large","title":"3D-LLM: Injecting the 3D World into Large Language Models","date":"2023-07-24","arxiv_id":"2307.12981","repositories_listed":5,"syntology":{"n":10,"n_ran":5,"n_unverified":5,"n_pointer_only":10}},{"url":"/paper/shapellm-universal-3d-object-understanding","title":"ShapeLLM: Universal 3D Object Understanding for Embodied Interaction","date":"2024-02-27","arxiv_id":"2402.17766","repositories_listed":3,"syntology":{"n":17,"n_ran":9,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","arxiv_id":"2311.17005","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/pointllm-empowering-large-language-models-to","title":"PointLLM: Empowering Large Language Models to Understand Point Clouds","date":"2023-08-31","arxiv_id":"2308.16911","repositories_listed":3,"syntology":null},{"url":"/paper/llava-onevision-easy-visual-task-transfer","title":"LLaVA-OneVision: Easy Visual Task Transfer","date":"2024-08-06","arxiv_id":"2408.03326","repositories_listed":2,"syntology":null},{"url":"/paper/chat-3d-v2-bridging-3d-scene-and-large","title":"Chat-Scene: Bridging 3D Scene and Large Language Models with Object Identifiers","date":"2023-12-13","arxiv_id":"2312.08168","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_unverified":0,"n_pointer_only":5}},{"url":"/paper/towards-learning-a-generalist-model-for","title":"Towards Learning a Generalist Model for Embodied Navigation","date":"2023-12-04","arxiv_id":"2312.02010","repositories_listed":2,"syntology":{"n":15,"n_ran":10,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/unveiling-the-mist-over-3d-vision-language","title":"Unveiling the Mist over 3D Vision-Language Understanding: Object-centric Evaluation with Chain-of-Analysis","date":"2025-03-28","arxiv_id":"2503.22420","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/dspnet-dual-vision-scene-perception-for","title":"DSPNet: Dual-vision Scene Perception for Robust 3D Question Answering","date":"2025-03-05","arxiv_id":"2503.03190","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/video-3d-llm-learning-position-aware-video","title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","date":"2024-11-30","arxiv_id":"2412.00493","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-situated-reasoning-in-3d-scenes","title":"Multi-modal Situated Reasoning in 3D Scenes","date":"2024-09-04","arxiv_id":"2409.02389","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","arxiv_id":"2402.15933","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":13}},{"url":"/paper/an-embodied-generalist-agent-in-3d-world","title":"An Embodied Generalist Agent in 3D World","date":"2023-11-18","arxiv_id":"2311.12871","repositories_listed":1,"syntology":{"n":13,"n_ran":3,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/scanqa-3d-question-answering-for-spatial","title":"ScanQA: 3D Question Answering for Spatial Scene Understanding","date":"2021-12-20","arxiv_id":"2112.10482","repositories_listed":1,"syntology":null}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}