{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/vision-and-language-navigation/papers/2","list_of":"/task/vision-and-language-navigation","task":"Vision and Language Navigation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":223,"counts":{"archive_papers_tagged":223,"with_a_code_link":114,"where_syntology_ran_a_sample":52,"not_listed_spam_title":0,"listed":223,"listed_where_code_ran":52,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":48,"every_run_a_failure_of_syntologys_instrument":4,"listed_with_a_run_with_no_instrument_failure":48,"listed_every_run_a_failure_of_syntologys_instrument":4,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/vision-and-language-navigation","prev":"/task/vision-and-language-navigation","next":"/task/vision-and-language-navigation/papers/3","papers":[{"url":"/paper/diagnosing-the-environment-bias-in-vision-and-1","slug":"diagnosing-the-environment-bias-in-vision-and-1","title":"Diagnosing the Environment Bias in Vision-and-Language Navigation","date":"2020-05-06","arxiv_id":"2005.03086","repositories_listed":1,"syntology":null},{"url":"/paper/improving-vision-and-language-navigation-with","slug":"improving-vision-and-language-navigation-with","title":"Improving Vision-and-Language Navigation with Image-Text Pairs from the Web","date":"2020-04-30","arxiv_id":"2004.14973","repositories_listed":1,"syntology":null},{"url":"/paper/sub-instruction-aware-vision-and-language","slug":"sub-instruction-aware-vision-and-language","title":"Sub-Instruction Aware Vision-and-Language Navigation","date":"2020-04-06","arxiv_id":"2004.02707","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sub-instruction-aware-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2004.02707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.02707"}},"official":{"repos":["YicongHong/Fine-Grained-R2R"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-learning-a-generic-agent-for-vision","slug":"towards-learning-a-generic-agent-for-vision","title":"Towards Learning a Generic Agent for Vision-and-Language Navigation via Pre-training","date":"2020-02-25","arxiv_id":"2002.10638","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-learning-a-generic-agent-for-vision#ran","syntology_url":"https://syntology.ai/paper/2002.10638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.10638"}},"official":{"repos":["weituo12321/PREVALENT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/valan-vision-and-language-agent-navigation","slug":"valan-vision-and-language-agent-navigation","title":"VALAN: Vision and Language Agent Navigation","date":"2019-12-06","arxiv_id":"1912.03241","repositories_listed":1,"syntology":null},{"url":"/paper/perceive-transform-and-act-multi-modal","slug":"perceive-transform-and-act-multi-modal","title":"Multimodal Attention Networks for Low-Level Vision-and-Language Navigation","date":"2019-11-27","arxiv_id":"1911.12377","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/perceive-transform-and-act-multi-modal#ran","syntology_url":"https://syntology.ai/paper/1911.12377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12377"}},"official":{"repos":["aimagelab/perceive-transform-and-act"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-navigation-with-language-pretraining","slug":"robust-navigation-with-language-pretraining","title":"Robust Navigation with Language Pretraining and Stochastic Sampling","date":"2019-09-05","arxiv_id":"1909.02244","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-vision-and-language-navigation-with","slug":"embodied-vision-and-language-navigation-with","title":"Embodied Vision-and-Language Navigation with Dynamic Convolutional Filters","date":"2019-07-05","arxiv_id":"1907.02985","repositories_listed":1,"syntology":null},{"url":"/paper/chasing-ghosts-instruction-following-as","slug":"chasing-ghosts-instruction-following-as","title":"Chasing Ghosts: Instruction Following as Bayesian State Tracking","date":"2019-07-03","arxiv_id":"1907.02022","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chasing-ghosts-instruction-following-as#ran","syntology_url":"https://syntology.ai/paper/1907.02022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.02022"}},"official":{"repos":["batra-mlp-lab/vln-chasing-ghosts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rerere-remote-embodied-referring-expressions","slug":"rerere-remote-embodied-referring-expressions","title":"REVERIE: Remote Embodied Visual Referring Expression in Real Indoor Environments","date":"2019-04-23","arxiv_id":"1904.10151","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rerere-remote-embodied-referring-expressions#ran","syntology_url":"https://syntology.ai/paper/1904.10151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10151"}},"official":null}},{"url":"/paper/tactical-rewind-self-correction-via","slug":"tactical-rewind-self-correction-via","title":"Tactical Rewind: Self-Correction via Backtracking in Vision-and-Language Navigation","date":"2019-03-06","arxiv_id":"1903.02547","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tactical-rewind-self-correction-via#ran","syntology_url":"https://syntology.ai/paper/1903.02547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.02547"}},"official":{"repos":["Kelym/FAST"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-regretful-navigation-agent-for-vision-and","slug":"the-regretful-navigation-agent-for-vision-and","title":"The Regretful Navigation Agent for Vision-and-Language Navigation","date":"2019-03-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/speaker-follower-models-for-vision-and","slug":"speaker-follower-models-for-vision-and","title":"Speaker-Follower Models for Vision-and-Language Navigation","date":"2018-06-07","arxiv_id":"1806.02724","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speaker-follower-models-for-vision-and#ran","syntology_url":"https://syntology.ai/paper/1806.02724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02724"}},"official":null}},{"url":"/paper/look-before-you-leap-bridging-model-free-and","slug":"look-before-you-leap-bridging-model-free-and","title":"Look Before You Leap: Bridging Model-Free and Model-Based Reinforcement Learning for Planned-Ahead Vision-and-Language Navigation","date":"2018-03-21","arxiv_id":"1803.07729","repositories_listed":1,"syntology":null},{"url":null,"slug":"rethinking-the-embodied-gap-in-vision-and","title":"Rethinking the Embodied Gap in Vision-and-Language Navigation: A Holistic Study of Physical and Visual Disparities","date":"2025-07-17","arxiv_id":"2507.13019","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounded-vision-language-navigation-for-uavs","title":"Grounded Vision-Language Navigation for UAVs with Open-Vocabulary Goal Understanding","date":"2025-06-12","arxiv_id":"2506.10756","repositories_listed":0,"syntology":null},{"url":null,"slug":"disrupting-vision-language-model-driven","title":"Disrupting Vision-Language Model-Driven Navigation Services via Adversarial Object Fusion","date":"2025-05-29","arxiv_id":"2505.23266","repositories_listed":0,"syntology":null},{"url":null,"slug":"metascenes-towards-automated-replica-creation","title":"MetaScenes: Towards Automated Replica Creation for Real-world 3D Scans","date":"2025-05-05","arxiv_id":"2505.02388","repositories_listed":0,"syntology":null},{"url":null,"slug":"dope-dual-object-perception-enhancement","title":"DOPE: Dual Object Perception-Enhancement Network for Vision-and-Language Navigation","date":"2025-04-30","arxiv_id":"2505.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"st-booster-an-iterative-spatiotemporal","title":"ST-Booster: An Iterative SpatioTemporal Perception Booster for Vision-and-Language Navigation in Continuous Environments","date":"2025-04-14","arxiv_id":"2504.09843","repositories_listed":0,"syntology":null},{"url":null,"slug":"endowing-embodied-agents-with-spatial","title":"Endowing Embodied Agents with Spatial Reasoning Capabilities for Vision-and-Language Navigation","date":"2025-04-09","arxiv_id":"2504.08806","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosmo-combination-of-selective-memorization","title":"COSMO: Combination of Selective Memorization for Low-cost Vision-and-Language Navigation","date":"2025-03-31","arxiv_id":"2503.24065","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-visual-imaginations-improve-vision-and","title":"Do Visual Imaginations Improve Vision-and-Language Navigation Agents?","date":"2025-03-20","arxiv_id":"2503.16394","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexvln-flexible-adaptation-for-diverse","title":"FlexVLN: Flexible Adaptation for Diverse Vision-and-Language Navigation Tasks","date":"2025-03-18","arxiv_id":"2503.13966","repositories_listed":0,"syntology":null},{"url":null,"slug":"ha-vln-a-benchmark-for-human-aware-navigation","title":"HA-VLN: A Benchmark for Human-Aware Navigation in Discrete-Continuous Environments with Dynamic Multi-Human Interactions, Real-World Validation, and an Open Leaderboard","date":"2025-03-18","arxiv_id":"2503.14229","repositories_listed":0,"syntology":null},{"url":null,"slug":"aerial-vision-and-language-navigation-with","title":"Aerial Vision-and-Language Navigation with Grid-based View Selection and Map Construction","date":"2025-03-14","arxiv_id":"2503.11091","repositories_listed":0,"syntology":null},{"url":null,"slug":"observation-graph-interaction-and-key-detail","title":"Observation-Graph Interaction and Key-Detail Guidance for Vision and Language Navigation","date":"2025-03-14","arxiv_id":"2503.11006","repositories_listed":0,"syntology":null},{"url":null,"slug":"panogen-domain-adapted-text-guided-panoramic","title":"PanoGen++: Domain-Adapted Text-Guided Panoramic Environment Generation for Vision-and-Language Navigation","date":"2025-03-13","arxiv_id":"2503.09938","repositories_listed":0,"syntology":null},{"url":null,"slug":"smartway-enhanced-waypoint-prediction-and","title":"SmartWay: Enhanced Waypoint Prediction and Backtracking for Zero-Shot Vision-and-Language Navigation","date":"2025-03-13","arxiv_id":"2503.10069","repositories_listed":0,"syntology":null},{"url":null,"slug":"ground-level-viewpoint-vision-and-language","title":"Ground-level Viewpoint Vision-and-Language Navigation in Continuous Environments","date":"2025-02-26","arxiv_id":"2502.19024","repositories_listed":0,"syntology":null},{"url":null,"slug":"travel-training-free-retrieval-and-alignment","title":"TRAVEL: Training-Free Retrieval and Alignment for Vision-and-Language Navigation","date":"2025-02-11","arxiv_id":"2502.07306","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-and-planning-in-robotic-navigation-a","title":"Language and Planning in Robotic Navigation: A Multilingual Evaluation of State-of-the-Art Models","date":"2025-01-07","arxiv_id":"2501.05478","repositories_listed":0,"syntology":null},{"url":null,"slug":"navcon-a-cognitively-inspired-and","title":"NAVCON: A Cognitively Inspired and Linguistically Grounded Corpus for Vision and Language Navigation","date":"2024-12-17","arxiv_id":"2412.13026","repositories_listed":0,"syntology":null},{"url":null,"slug":"roomtour3d-geometry-aware-video-instruction","title":"RoomTour3D: Geometry-Aware Video-Instruction Tuning for Embodied Navigation","date":"2024-12-11","arxiv_id":"2412.08591","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-consistent-data-generation-for-vision","title":"World-Consistent Data Generation for Vision-and-Language Navigation","date":"2024-12-09","arxiv_id":"2412.06413","repositories_listed":0,"syntology":null},{"url":null,"slug":"navila-legged-robot-vision-language-action","title":"NaVILA: Legged Robot Vision-Language-Action Model for Navigation","date":"2024-12-05","arxiv_id":"2412.04453","repositories_listed":0,"syntology":null},{"url":null,"slug":"hijacking-vision-and-language-navigation","title":"Hijacking Vision-and-Language Navigation Agents with Adversarial Environmental Attacks","date":"2024-12-03","arxiv_id":"2412.02795","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-from-imagination-episodic-simulation","title":"Planning from Imagination: Episodic Simulation and Episodic Memory for Vision-and-Language Navigation","date":"2024-11-30","arxiv_id":"2412.01857","repositories_listed":0,"syntology":null},{"url":null,"slug":"unitedvln-generalizable-gaussian-splatting","title":"UnitedVLN: Generalizable Gaussian Splatting for Continuous Vision-Language Navigation","date":"2024-11-25","arxiv_id":"2411.16053","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-alignment-in-vision-and-language","title":"Fine-Grained Alignment in Vision-and-Language Navigation through Bayesian Optimization","date":"2024-11-22","arxiv_id":"2411.14811","repositories_listed":0,"syntology":null},{"url":null,"slug":"navagent-multi-scale-urban-street-view-fusion","title":"NavAgent: Multi-scale Urban Street View Fusion For UAV Embodied Vision-and-Language Navigation","date":"2024-11-13","arxiv_id":"2411.08579","repositories_listed":0,"syntology":null},{"url":null,"slug":"aerial-vision-and-language-navigation-via","title":"Aerial Vision-and-Language Navigation via Semantic-Topo-Metric Representation Guided LLM Reasoning","date":"2024-10-11","arxiv_id":"2410.08500","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-vision-and-language-navigation-with","title":"Zero-Shot Vision-and-Language Navigation with Collision Mitigation in Continuous Environment","date":"2024-10-07","arxiv_id":"2410.17267","repositories_listed":0,"syntology":null},{"url":null,"slug":"minivln-efficient-vision-and-language","title":"MiniVLN: Efficient Vision-and-Language Navigation by Progressive Knowledge Distillation","date":"2024-09-27","arxiv_id":"2409.18800","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-nav-exploring-zero-shot-vision-and","title":"Open-Nav: Exploring Zero-Shot Vision-and-Language Navigation in Continuous Environment with Open-Source LLMs","date":"2024-09-27","arxiv_id":"2409.18794","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-is-believing-enhancing-vision-language","title":"Seeing is Believing? Enhancing Vision-Language Navigation using Visual Perturbations","date":"2024-09-09","arxiv_id":"2409.05552","repositories_listed":0,"syntology":null},{"url":null,"slug":"loc4plan-locating-before-planning-for-outdoor","title":"Loc4Plan: Locating Before Planning for Outdoor Vision and Language Navigation","date":"2024-08-09","arxiv_id":"2408.05090","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrast-sets-for-evaluating-language-guided","title":"Contrast Sets for Evaluating Language-Guided Robot Policies","date":"2024-06-19","arxiv_id":"2406.13636","repositories_listed":0,"syntology":null},{"url":null,"slug":"i2edl-interactive-instruction-error-detection","title":"I2EDL: Interactive Instruction Error Detection and Localization","date":"2024-06-07","arxiv_id":"2406.05080","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-and-language-navigation-generative","title":"Vision-and-Language Navigation Generative Pretrained Transformer","date":"2024-05-27","arxiv_id":"2405.16994","repositories_listed":0,"syntology":null},{"url":null,"slug":"mc-gpt-empowering-vision-and-language","title":"MC-GPT: Empowering Vision-and-Language Navigation with Memory Map and Reasoning Chains","date":"2024-05-17","arxiv_id":"2405.10620","repositories_listed":0,"syntology":null},{"url":null,"slug":"aigen-an-adversarial-approach-for-instruction","title":"AIGeN: An Adversarial Approach for Instruction Generation in VLN","date":"2024-04-15","arxiv_id":"2404.10054","repositories_listed":0,"syntology":null},{"url":null,"slug":"ivlmap-instance-aware-visual-language","title":"IVLMap: Instance-Aware Visual Language Grounding for Consumer Robot Navigation","date":"2024-03-28","arxiv_id":"2403.19336","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-vision-and-language-navigation-with","title":"Scaling Vision-and-Language Navigation With Offline RL","date":"2024-03-27","arxiv_id":"2403.18454","repositories_listed":0,"syntology":null},{"url":null,"slug":"over-nav-elevating-iterative-vision-and","title":"OVER-NAV: Elevating Iterative Vision-and-Language Navigation with Open-Vocabulary Detection and StructurEd Representation","date":"2024-03-26","arxiv_id":"2403.17334","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-spatial-object-relations-modeling","title":"Temporal-Spatial Object Relations Modeling for Vision-and-Language Navigation","date":"2024-03-23","arxiv_id":"2403.15691","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-vision-and-language-navigation","title":"Continual Vision-and-Language Navigation","date":"2024-03-22","arxiv_id":"2403.15049","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-error-detection-and-localization-of","title":"Mind the Error! Detection and Localization of Instruction Errors in Vision-and-Language Navigation","date":"2024-03-15","arxiv_id":"2403.10700","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-deviation-robust-agent-navigation-via","title":"Towards Deviation-Robust Agent Navigation via Perturbation-Aware Contrastive Learning","date":"2024-03-09","arxiv_id":"2403.05770","repositories_listed":0,"syntology":null},{"url":null,"slug":"causality-based-cross-modal-representation","title":"Causality-based Cross-Modal Representation Learning for Vision-and-Language Navigation","date":"2024-03-06","arxiv_id":"2403.03405","repositories_listed":0,"syntology":null},{"url":null,"slug":"navid-video-based-vlm-plans-the-next-step-for","title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","date":"2024-02-24","arxiv_id":"2402.15852","repositories_listed":0,"syntology":null},{"url":null,"slug":"vln-video-utilizing-driving-videos-for","title":"VLN-Video: Utilizing Driving Videos for Outdoor Vision-and-Language Navigation","date":"2024-02-05","arxiv_id":"2402.03561","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapgpt-map-guided-prompting-for-unified","title":"MapGPT: Map-Guided Prompting with Adaptive Path Planning for Vision-and-Language Navigation","date":"2024-01-14","arxiv_id":"2401.07314","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-way-is-right-uncovering-limitations-of","title":"Which way is `right'?: Uncovering limitations of Vision-and-Language Navigation model","date":"2023-11-30","arxiv_id":"2312.00151","repositories_listed":0,"syntology":null},{"url":null,"slug":"dap-domain-aware-prompt-learning-for-vision","title":"DAP: Domain-aware Prompt Learning for Vision-and-Language Navigation","date":"2023-11-29","arxiv_id":"2311.17812","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-vln-pretraining-work-with-nonsensical-or","title":"Does VLN Pretraining Work with Nonsensical or Irrelevant Instructions?","date":"2023-11-28","arxiv_id":"2311.17280","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-and-language-navigation-in-the-real","title":"Vision and Language Navigation in the Real World via Online Visual Language Mapping","date":"2023-10-16","arxiv_id":"2310.10822","repositories_listed":0,"syntology":null},{"url":null,"slug":"langnav-language-as-a-perceptual","title":"LangNav: Language as a Perceptual Representation for Navigation","date":"2023-10-11","arxiv_id":"2310.07889","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-explanation-methods-for-vision-and","title":"Evaluating Explanation Methods for Vision-and-Language Navigation","date":"2023-10-10","arxiv_id":"2310.06654","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-based-context-and-domain-aware","title":"Prompt-based Context- and Domain-aware Pretraining for Vision and Language Navigation","date":"2023-09-07","arxiv_id":"2309.03661","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-2-nav-action-aware-zero-shot-robot","title":"$A^2$Nav: Action-Aware Zero-Shot Robot Navigation by Exploiting Vision-and-Language Ability of Foundation Models","date":"2023-08-15","arxiv_id":"2308.07997","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-gap-improving-success-rate-of-vision","title":"Mind the Gap: Improving Success Rate of Vision-and-Language Navigation by Revisiting Oracle Success Routes","date":"2023-08-07","arxiv_id":"2308.03244","repositories_listed":0,"syntology":null},{"url":null,"slug":"mo-vln-a-multi-task-benchmark-for-open-set","title":"CorNav: Autonomous Agent with Self-Corrected Planning for Zero-Shot Vision-and-Language Navigation","date":"2023-06-17","arxiv_id":"2306.10322","repositories_listed":0,"syntology":null},{"url":null,"slug":"panogen-text-conditioned-panoramic","title":"PanoGen: Text-Conditioned Panoramic Environment Generation for Vision-and-Language Navigation","date":"2023-05-30","arxiv_id":"2305.19195","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-path-modeling-for-vision-and-language","title":"Masked Path Modeling for Vision-and-Language Navigation","date":"2023-05-23","arxiv_id":"2305.14268","repositories_listed":0,"syntology":null},{"url":null,"slug":"pasts-progress-aware-spatio-temporal","title":"PASTS: Progress-Aware Spatio-Temporal Transformer Speaker For Vision-and-Language Navigation","date":"2023-05-19","arxiv_id":"2305.11918","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-vision-and-language-navigation-by","title":"Improving Vision-and-Language Navigation by Generating Future-View Image Semantics","date":"2023-04-11","arxiv_id":"2304.04907","repositories_listed":0,"syntology":null},{"url":null,"slug":"hop-history-enhanced-and-order-aware-pre","title":"HOP+: History-enhanced and Order-aware Pre-training for Vision-and-Language Navigation","date":"2023-03-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/meta-explore-exploratory-hierarchical-vision","slug":"meta-explore-exploratory-hierarchical-vision","title":"Meta-Explore: Exploratory Hierarchical Vision-and-Language Navigation Using Scene Object Spectrum Grounding","date":"2023-03-07","arxiv_id":"2303.04077","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-environment-representation-for","title":"Graph based Environment Representation for Vision-and-Language Navigation in Continuous Environments","date":"2023-01-11","arxiv_id":"2301.04352","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-nav-using-clip-for-zero-shot-vision-and","title":"CLIP-Nav: Using CLIP for Zero-Shot Vision-and-Language Navigation","date":"2022-11-30","arxiv_id":"2211.16649","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-as-the-attacker-wishes-towards","title":"Navigation as Attackers Wish? Towards Building Robust Embodied Agents under Federated Learning","date":"2022-11-27","arxiv_id":"2211.14769","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-encoding-auxiliary-tasks-for","title":"Structure-Encoding Auxiliary Tasks for Improved Visual Representation in Vision-and-Language Navigation","date":"2022-11-20","arxiv_id":"2211.11116","repositories_listed":0,"syntology":null},{"url":"/paper/a-new-path-scaling-vision-and-language","slug":"a-new-path-scaling-vision-and-language","title":"A New Path: Scaling Vision-and-Language Navigation with Synthetic Instructions and Imitation Learning","date":"2022-10-06","arxiv_id":"2210.03112","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-vision-and-language-navigation","title":"Iterative Vision-and-Language Navigation","date":"2022-10-06","arxiv_id":"2210.03087","repositories_listed":0,"syntology":null},{"url":null,"slug":"anticipating-the-unseen-discrepancy-for","title":"Anticipating the Unseen Discrepancy for Vision and Language Navigation","date":"2022-09-10","arxiv_id":"2209.04725","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-2-sim-transfer-for-vision-and-language","title":"Sim-2-Sim Transfer for Vision-and-Language Navigation in Continuous Environments","date":"2022-04-20","arxiv_id":"2204.09667","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-3d-semantic-representation","title":"Self-supervised 3D Semantic Representation Learning for Vision-and-Language Navigation","date":"2022-01-26","arxiv_id":"2201.10788","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-the-potential-performance-of-vision-1","title":"Explore the Potential Performance of Vision-and-Language Navigation Model: a Snapshot Ensemble Method","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"diagnosing-vision-and-language-navigation-1","title":"Diagnosing Vision-and-Language Navigation: What Really Matters","date":"2021-12-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-the-potential-performance-of-vision","title":"Explore the Potential Performance of Vision-and-Language Navigation Model: a Snapshot Ensemble Method","date":"2021-11-28","arxiv_id":"2111.14267","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-object-relation-alignment-for-vision","title":"Explicit Object Relation Alignment for Vision and Language Navigation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-and-language-navigation-a-survey-of","title":"Vision-and-Language Navigation: A Survey of Tasks, Methods, and Future Directions","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-for-vision-and-language","title":"Curriculum Learning for Vision-and-Language Navigation","date":"2021-11-14","arxiv_id":"2111.07228","repositories_listed":0,"syntology":null},{"url":null,"slug":"soat-a-scene-and-object-aware-transformer-for","title":"SOAT: A Scene- and Object-Aware Transformer for Vision-and-Language Navigation","date":"2021-10-27","arxiv_id":"2110.14143","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-spatial-route-prior-in-vision","title":"Rethinking the Spatial Route Prior in Vision-and-Language Navigation","date":"2021-10-12","arxiv_id":"2110.05728","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-aligned-waypoint-law-supervision-for","title":"Language-Aligned Waypoint (LAW) Supervision for Vision-and-Language Navigation in Continuous Environments","date":"2021-09-30","arxiv_id":"2109.15207","repositories_listed":0,"syntology":null},{"url":null,"slug":"vln-bert-a-recurrent-vision-and-language-bert","title":"VLN BERT: A Recurrent Vision-and-Language BERT for Navigation","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"crossmap-transformer-a-crossmodal-masked-path","title":"CrossMap Transformer: A Crossmodal Masked Path Transformer Using Double Back-Translation for Vision-and-Language Navigation","date":"2021-03-01","arxiv_id":"2103.00852","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-evaluation-of-vision-and-language","title":"On the Evaluation of Vision-and-Language Navigation Instructions","date":"2021-01-26","arxiv_id":"2101.10504","repositories_listed":0,"syntology":null}],"record_sha256":"14fed052132af93f9eb235d9dafa3ad3e9da9cb890ecfac107a660ffcdd56a5e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}