{"url":"/dataset/ai2-thor","name":"AI2-THOR","full_name":"AI2-THOR","description_markdown":"AI2-Thor is an interactive environment for embodied AI. It contains four types of scenes, including kitchen, living room, bedroom and bathroom, and each scene includes 30 rooms, where each room is unique in terms of furniture placement and item types. There are over 2000 unique objects for AI agents to interact with.\r\n\r\nSource: [Learning Object Relation Graph andTentative Policy for Visual Navigation](https://arxiv.org/abs/2007.11018)\r\nImage Source: [https://ai2thor.allenai.org/](https://ai2thor.allenai.org/)","description_withheld":null,"homepage":"https://ai2thor.allenai.org/","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/ai2-thor-an-interactive-3d-environment-for","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","first_author":"Eric Kolve","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Environment","url":"/datasets/modality/environment"}],"tasks":[{"name":"Visual Navigation","url":"/task/visual-navigation","datasets_with_task":"/datasets/task/visual-navigation"},{"name":"Imitation Learning","url":"/task/imitation-learning","datasets_with_task":"/datasets/task/imitation-learning"}],"languages":[],"variants":["AI2-THOR"],"data_loaders":[],"num_papers_in_archive":243,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-navigation-on-ai2-thor","task":"Visual Navigation","dataset_variant":"AI2-THOR","rows":2,"metrics":["SPL (All)","SPL (L≥5)","Success Rate (All)","Success Rate (L≥5)"],"first_row_in_archive_order":{"model":"MVV-IN","paper":"/paper/multimodal-aggregation-approach-for-memory","metrics":{"SPL (All)":"17.27","SPL (L≥5)":"13.63","Success Rate (All)":"48.7","Success Rate (L≥5)":"30.9"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/multimodal-aggregation-approach-for-memory","title":"Multimodal Aggregation Approach for Memory Vision-Voice Indoor Navigation with Meta-Learning","date":"2020-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-to-learn-how-to-learn-self-adaptive","title":"Learning to Learn How to Learn: Self-Adaptive Visual Navigation Using Meta-Learning","date":"2018-12-03","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}