{"url":"/dataset/scanrefer-dataset","name":"ScanRefer Dataset","full_name":null,"description_markdown":"Contains 51,583 descriptions of 11,046 objects from 800 ScanNet scenes. ScanRefer is the first large-scale effort to perform object localization via natural language expression directly in 3D.\r\n\r\nSource: [ScanRefer: 3D Object Localization in RGB-D Scans using Natural Language](/paper/scanrefer-3d-object-localization-in-rgb-d)","description_withheld":null,"homepage":"https://github.com/daveredrum/ScanRefer","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/scanrefer-3d-object-localization-in-rgb-d","title":"ScanRefer: 3D Object Localization in RGB-D Scans using Natural Language","first_author":"Dave Zhenyu Chen","url":null},"license":null,"modalities":[],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Object Localization","url":"/task/object-localization","datasets_with_task":"/datasets/task/object-localization"},{"name":"Sentence Embeddings","url":"/task/sentence-embeddings","datasets_with_task":"/datasets/task/sentence-embeddings"},{"name":"3D dense captioning","url":"/task/3d-dense-captioning","datasets_with_task":"/datasets/task/3d-dense-captioning"}],"languages":[],"variants":["ScanRefer Dataset"],"data_loaders":[{"repo":"https://github.com/daveredrum/ScanRefer","url":"https://github.com/daveredrum/ScanRefer","frameworks":["pytorch"]}],"num_papers_in_archive":63,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/3d-dense-captioning-on-scanrefer-dataset","task":"3D dense captioning","dataset_variant":"ScanRefer Dataset","rows":12,"metrics":["CIDEr","BLEU-4","METEOR","ROUGE-L"],"first_row_in_archive_order":{"model":"3D CoCa","paper":"/paper/3d-coca-contrastive-learners-are-3d","metrics":{"BLEU-4":"45.56","CIDEr":"85.42","METEOR":"30.95","ROUGE-L":"61.98"},"code_links":[{"title":"AIGeeksGroup/3DCoCa","url":"https://github.com/AIGeeksGroup/3DCoCa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/3d-coca-contrastive-learners-are-3d","title":"3D CoCa: Contrastive Learners are 3D Captioners","date":"2025-04-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/see-it-all-contextualized-late-aggregation","title":"See It All: Contextualized Late Aggregation for 3D Dense Captioning","date":"2024-08-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bi-directional-contextual-attention-for-3d","title":"Bi-directional Contextual Attention for 3D Dense Captioning","date":"2024-08-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vote2cap-detr-decoupling-localization-and","title":"Vote2Cap-DETR++: Decoupling Localization and Describing for End-to-End 3D Dense Captioning","date":"2023-09-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-3d-dense-captioning-with-vote2cap","title":"End-to-End 3D Dense Captioning with Vote2Cap-DETR","date":"2023-01-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-aware-alignment-and-mutual-masking","title":"Context-Aware Alignment and Mutual Masking for 3D-Language Pre-Training","date":"2023-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/contextual-modeling-for-3d-dense-captioning","title":"Contextual Modeling for 3D Dense Captioning on Point Clouds","date":"2022-10-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spatiality-guided-transformer-for-3d-dense","title":"Spatiality-guided Transformer for 3D Dense Captioning on Point Clouds","date":"2022-04-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/more-multi-order-relation-mining-for-dense","title":"MORE: Multi-Order RElation Mining for Dense Captioning in 3D Scenes","date":"2022-03-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/x-trans2cap-cross-modal-knowledge-transfer","title":"X-Trans2Cap: Cross-Modal Knowledge Transfer using Transformer for 3D Dense Captioning","date":"2022-03-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/3djcg-a-unified-framework-for-joint-dense","title":"3DJCG: A Unified Framework for Joint Dense Captioning and Visual Grounding on 3D Point Clouds","date":"2022-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scan2cap-context-aware-dense-captioning-in","title":"Scan2Cap: Context-aware Dense Captioning in RGB-D Scans","date":"2020-12-03","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":12,"samples_ran":3,"samples_unverified":9,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}