{"url":"/task/vision-and-language-navigation","name":"Vision and Language Navigation","slug":"vision-and-language-navigation","description_markdown":null,"categories":[{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":223,"papers_with_code":114,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":13,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/vision-and-language-navigation-on-vln","slug":"vision-and-language-navigation-on-vln","dataset":"VLN Challenge","dataset_url":null,"rows_in_archive":145,"metrics":["success","length","error","oracle success","spl"],"first_row_in_archive_order":{"model":"human","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/vision-and-language-navigation-on-touchdown","slug":"vision-and-language-navigation-on-touchdown","dataset":"Touchdown Dataset","dataset_url":"/dataset/touchdown-dataset","rows_in_archive":12,"metrics":["Task Completion (TC)"],"first_row_in_archive_order":{"model":"FLAME","paper_title":"FLAME: Learning to Navigate with Multimodal LLM in Urban Environments","paper_url":"/paper/flame-learning-to-navigate-with-multimodal","paper_date":"2024-08-20","arxiv_id":"2408.11051","code_links":[{"title":"xyz9911/FLAME","url":"https://github.com/xyz9911/FLAME"}],"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/vision-and-language-navigation-on-rxr","slug":"vision-and-language-navigation-on-rxr","dataset":"RxR","dataset_url":"/dataset/rxr","rows_in_archive":6,"metrics":["ndtw"],"first_row_in_archive_order":{"model":"MARVAL","paper_title":"A New Path: Scaling Vision-and-Language Navigation with Synthetic Instructions and Imitation Learning","paper_url":"/paper/a-new-path-scaling-vision-and-language","paper_date":"2022-10-06","arxiv_id":"2210.03112","code_links":[],"syntology":null}},{"leaderboard":"/sota/vision-and-language-navigation-on-map2seq","slug":"vision-and-language-navigation-on-map2seq","dataset":"map2seq","dataset_url":"/dataset/map2seq","rows_in_archive":5,"metrics":["Task Completion (TC)"],"first_row_in_archive_order":{"model":"FLAME","paper_title":"FLAME: Learning to Navigate with Multimodal LLM in Urban Environments","paper_url":"/paper/flame-learning-to-navigate-with-multimodal","paper_date":"2024-08-20","arxiv_id":"2408.11051","code_links":[{"title":"xyz9911/FLAME","url":"https://github.com/xyz9911/FLAME"}],"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/vision-and-language-navigation-on-robo-vln","slug":"vision-and-language-navigation-on-robo-vln","dataset":"robo-vln","dataset_url":"/dataset/robo-vln","rows_in_archive":1,"metrics":["SPL (Sucess Weighted by Path Length)"],"first_row_in_archive_order":{"model":"Hierarchical Cross-Modal Agent","paper_title":"Hierarchical Cross-Modal Agent for Robotics Vision-and-Language Navigation","paper_url":"/paper/hierarchical-cross-modal-agent-for-robotics","paper_date":"2021-04-21","arxiv_id":"2104.10674","code_links":[{"title":"GT-RIPL/robo-vln","url":"https://github.com/GT-RIPL/robo-vln"}],"syntology":null}}],"datasets":[{"url":"/dataset/rxr","name":"RxR","full_name":"Room-across-Room","num_papers_in_archive":54},{"url":"/dataset/streetlearn","name":"StreetLearn","full_name":"","num_papers_in_archive":30},{"url":"/dataset/minos","name":"MINOS","full_name":"MINOS","num_papers_in_archive":22},{"url":"/dataset/touchdown-dataset","name":"Touchdown Dataset","full_name":null,"num_papers_in_archive":19},{"url":"/dataset/talk-the-walk","name":"Talk the Walk","full_name":"","num_papers_in_archive":11},{"url":"/dataset/chalet","name":"CHALET","full_name":"Cornell House Agent Learning Environment","num_papers_in_archive":10},{"url":"/dataset/weblinx","name":"WebLINX","full_name":"Real-World Website Navigation with Multi-Turn","num_papers_in_archive":6},{"url":"/dataset/map2seq","name":"map2seq","full_name":"","num_papers_in_archive":5},{"url":"/dataset/run","name":"RUN","full_name":"The RUN Dataset","num_papers_in_archive":5},{"url":"/dataset/arramon","name":"ArraMon","full_name":"","num_papers_in_archive":3},{"url":"/dataset/fine-grained-r2r","name":"Fine-Grained R2R","full_name":null,"num_papers_in_archive":2},{"url":"/dataset/robo-vln","name":"robo-vln","full_name":"Robotics Vision-and-Language Navigation","num_papers_in_archive":2},{"url":"/dataset/talk2nav","name":"Talk2Nav","full_name":"","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":114,"tagged_in_all":223,"items":[{"url":"/paper/vision-and-language-navigation-interpreting","title":"Vision-and-Language Navigation: Interpreting visually-grounded navigation instructions in real environments","date":"2017-11-20","arxiv_id":"1711.07280","repositories_listed":8,"syntology":null},{"url":"/paper/beyond-the-nav-graph-vision-and-language","title":"Beyond the Nav-Graph: Vision-and-Language Navigation in Continuous Environments","date":"2020-04-06","arxiv_id":"2004.02857","repositories_listed":5,"syntology":null},{"url":"/paper/how-much-can-clip-benefit-vision-and-language","title":"How Much Can CLIP Benefit Vision-and-Language Tasks?","date":"2021-07-13","arxiv_id":"2107.06383","repositories_listed":4,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":9}},{"url":"/paper/retouchdown-adding-touchdown-to-streetlearn","title":"Retouchdown: Adding Touchdown to StreetLearn as a Shareable Resource for Language Grounding Tasks in Street View","date":"2020-01-10","arxiv_id":"2001.03671","repositories_listed":4,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/touchdown-natural-language-navigation-and","title":"Touchdown: Natural Language Navigation and Spatial Reasoning in Visual Street Environments","date":"2018-11-29","arxiv_id":"1811.12354","repositories_listed":4,"syntology":null},{"url":"/paper/room-across-room-multilingual-vision-and","title":"Room-Across-Room: Multilingual Vision-and-Language Navigation with Dense Spatiotemporal Grounding","date":"2020-10-15","arxiv_id":"2010.07954","repositories_listed":3,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/the-regretful-agent-heuristic-aided","title":"The Regretful Agent: Heuristic-Aided Navigation through Progress Estimation","date":"2019-03-05","arxiv_id":"1903.01602","repositories_listed":3,"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/hierarchical-spatial-proximity-reasoning-for","title":"Hierarchical Spatial Proximity Reasoning for Vision-and-Language Navigation","date":"2024-03-18","arxiv_id":"2403.11541","repositories_listed":2,"syntology":null},{"url":"/paper/weblinx-real-world-website-navigation-with","title":"WebLINX: Real-World Website Navigation with Multi-Turn Dialogue","date":"2024-02-08","arxiv_id":"2402.05930","repositories_listed":2,"syntology":{"n":27,"n_ran":21,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/navgpt-explicit-reasoning-in-vision-and","title":"NavGPT: Explicit Reasoning in Vision-and-Language Navigation with Large Language Models","date":"2023-05-26","arxiv_id":"2305.16986","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/airbert-in-domain-pretraining-for-vision-and","title":"Airbert: In-domain Pretraining for Vision-and-Language Navigation","date":"2021-08-20","arxiv_id":"2108.09105","repositories_listed":2,"syntology":{"n":14,"n_ran":7,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/self-monitoring-navigation-agent-via","title":"Self-Monitoring Navigation Agent via Auxiliary Progress Estimation","date":"2019-01-10","arxiv_id":"1901.03035","repositories_listed":2,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/navmorph-a-self-evolving-world-model-for","title":"NavMorph: A Self-Evolving World Model for Vision-and-Language Navigation in Continuous Environments","date":"2025-06-30","arxiv_id":"2506.23468","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":12}},{"url":"/paper/2506-10172","title":"A Navigation Framework Utilizing Vision-Language Models","date":"2025-06-11","arxiv_id":"2506.10172","repositories_listed":1,"syntology":null},{"url":"/paper/cross-from-left-to-right-brain-adaptive-text","title":"Cross from Left to Right Brain: Adaptive Text Dreamer for Vision-and-Language Navigation","date":"2025-05-27","arxiv_id":"2505.20897","repositories_listed":1,"syntology":null},{"url":"/paper/flightgpt-towards-generalizable-and","title":"FlightGPT: Towards Generalizable and Interpretable UAV Vision-and-Language Navigation with Vision-Language Models","date":"2025-05-19","arxiv_id":"2505.12835","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11383","title":"Dynam3D: Dynamic Layered 3D Tokens Empower VLM for Vision-and-Language Navigation","date":"2025-05-16","arxiv_id":"2505.11383","repositories_listed":1,"syntology":null},{"url":"/paper/citynavagent-aerial-vision-and-language","title":"CityNavAgent: Aerial Vision-and-Language Navigation with Hierarchical Semantic Planning and Global Memory","date":"2025-05-08","arxiv_id":"2505.05622","repositories_listed":1,"syntology":null},{"url":"/paper/navrag-generating-user-demand-instructions","title":"NavRAG: Generating User Demand Instructions for Embodied Navigation through Retrieval-Augmented LLM","date":"2025-02-16","arxiv_id":"2502.11142","repositories_listed":1,"syntology":null},{"url":"/paper/general-scene-adaptation-for-vision-and","title":"General Scene Adaptation for Vision-and-Language Navigation","date":"2025-01-29","arxiv_id":"2501.17403","repositories_listed":1,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/agent-journey-beyond-rgb-unveiling-hybrid","title":"Agent Journey Beyond RGB: Unveiling Hybrid Semantic-Spatial Environmental Representations for Vision-and-Language Navigation","date":"2024-12-09","arxiv_id":"2412.06465","repositories_listed":1,"syntology":null},{"url":"/paper/g3d-lf-generalizable-3d-language-feature","title":"g3D-LF: Generalizable 3D-Language Feature Fields for Embodied Tasks","date":"2024-11-26","arxiv_id":"2411.17030","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/spatially-aware-speaker-for-vision-and","title":"Spatially-Aware Speaker for Vision-and-Language Navigation Instruction Generation","date":"2024-09-09","arxiv_id":"2409.05583","repositories_listed":1,"syntology":null},{"url":"/paper/flame-learning-to-navigate-with-multimodal","title":"FLAME: Learning to Navigate with Multimodal LLM in Urban Environments","date":"2024-08-20","arxiv_id":"2408.11051","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/narrowing-the-gap-between-vision-and-action","title":"Narrowing the Gap between Vision and Action in Navigation","date":"2024-08-19","arxiv_id":"2408.10388","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21452","title":"Navigating Beyond Instructions: Vision-and-Language Navigation in Obstructed Environments","date":"2024-07-31","arxiv_id":"2407.21452","repositories_listed":1,"syntology":null},{"url":"/paper/navgpt-2-unleashing-navigational-reasoning","title":"NavGPT-2: Unleashing Navigational Reasoning Capability for Large Vision-Language Models","date":"2024-07-17","arxiv_id":"2407.12366","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/pret-planning-with-directed-fidelity","title":"PRET: Planning with Directed Fidelity Trajectory for Vision and Language Navigation","date":"2024-07-16","arxiv_id":"2407.11487","repositories_listed":1,"syntology":null},{"url":"/paper/vision-and-language-navigation-today-and","title":"Vision-and-Language Navigation Today and Tomorrow: A Survey in the Era of Foundation Models","date":"2024-07-09","arxiv_id":"2407.07035","repositories_listed":1,"syntology":null},{"url":"/paper/into-the-unknown-generating-geospatial","title":"Into the Unknown: Generating Geospatial Descriptions for New Environments","date":"2024-06-28","arxiv_id":"2406.19967","repositories_listed":1,"syntology":null}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}