{"url":"/dataset/room-to-room","name":"R2R","full_name":"Room-to-Room","description_markdown":"R2R is a dataset for visually-grounded natural language navigation in real buildings. The dataset requires autonomous agents to follow human-generated navigation instructions in previously unseen buildings, as illustrated in the demo above. For training, each instruction is associated with a Matterport3D Simulator trajectory. 22k instructions are available, with an average length of 29 words. There is a test evaluation server for this dataset available at EvalAI.\r\n\r\nSource: [Natural language interaction with robots](https://bringmeaspoon.org/)","description_withheld":null,"homepage":"https://bringmeaspoon.org/","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/vision-and-language-navigation-interpreting","title":"Vision-and-Language Navigation: Interpreting visually-grounded navigation instructions in real environments","first_author":"Peter Anderson","url":null},"license":{"name":"Custom (research-only)","url":"https://bringmeaspoon.org/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Interactive","url":"/datasets/modality/interactive"}],"tasks":[{"name":"Visual Navigation","url":"/task/visual-navigation","datasets_with_task":"/datasets/task/visual-navigation"},{"name":"Vision-Language Navigation","url":"/task/vision-language-navigation","datasets_with_task":"/datasets/task/vision-language-navigation"}],"languages":[],"variants":["Room2Room","R2R"],"data_loaders":[],"num_papers_in_archive":174,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-navigation-on-room-to-room-1","task":"Visual Navigation","dataset_variant":"R2R","rows":11,"metrics":["spl"],"first_row_in_archive_order":{"model":"SUSA","paper":"/paper/agent-journey-beyond-rgb-unveiling-hybrid","metrics":{"spl":"0.6383"},"code_links":[{"title":"HCI-LMC/VLN-SUSA","url":"https://github.com/HCI-LMC/VLN-SUSA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/vision-language-navigation-on-room2room","task":"Vision-Language Navigation","dataset_variant":"Room2Room","rows":3,"metrics":["spl"],"first_row_in_archive_order":{"model":"R2R+EnvDrop","paper":"/paper/learning-to-navigate-unseen-environments-back","metrics":{"spl":"0.61"},"code_links":[{"title":"airsplay/R2R-EnvDrop","url":"https://github.com/airsplay/R2R-EnvDrop"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/agent-journey-beyond-rgb-unveiling-hybrid","title":"Agent Journey Beyond RGB: Unveiling Hybrid Semantic-Spatial Environmental Representations for Vision-and-Language Navigation","date":"2024-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/towards-learning-a-generalist-model-for","title":"Towards Learning a Generalist Model for Embodied Navigation","date":"2023-12-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vln-petl-parameter-efficient-transfer","title":"VLN-PETL: Parameter-Efficient Transfer Learning for Vision-and-Language Navigation","date":"2023-08-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/meta-explore-exploratory-hierarchical-vision","title":"Meta-Explore: Exploratory Hierarchical Vision-and-Language Navigation Using Scene Object Spectrum Grounding","date":"2023-03-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bevbert-topo-metric-map-pre-training-for","title":"BEVBert: Multimodal Map Pre-training for Language-guided Navigation","date":"2022-12-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hop-history-and-order-aware-pre-training-for","title":"HOP: History-and-Order Aware Pre-training for Vision-and-Language Navigation","date":"2022-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":4,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/think-global-act-local-dual-scale-graph","title":"Think Global, Act Local: Dual-scale Graph Transformer for Vision-and-Language Navigation","date":"2022-02-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-recurrent-vision-and-language-bert-for","title":"A Recurrent Vision-and-Language BERT for Navigation","date":"2020-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/towards-learning-a-generic-agent-for-vision","title":"Towards Learning a Generic Agent for Vision-and-Language Navigation via Pre-training","date":"2020-02-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-navigate-unseen-environments-back","title":"Learning to Navigate Unseen Environments: Back Translation with Environmental Dropout","date":"2019-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tactical-rewind-self-correction-via","title":"Tactical Rewind: Self-Correction via Backtracking in Vision-and-Language Navigation","date":"2019-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reinforced-cross-modal-matching-and-self","title":"Reinforced Cross-Modal Matching and Self-Supervised Imitation Learning for Vision-Language Navigation","date":"2018-11-25","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/vision-and-language-navigation-interpreting","title":"Vision-and-Language Navigation: Interpreting visually-grounded navigation instructions in real environments","date":"2017-11-20","rows_on_this_dataset":1,"code_links":8,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":41,"samples_ran":22,"samples_unverified":19,"pointer_only_for_licence":9,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}