{"url":"/task/visual-localization","name":"Visual Localization","slug":"visual-localization","description_markdown":"**Visual Localization** is the problem of estimating the camera pose of a given image relative to a visual representation of a known scene.\n\n\n<span class=\"description-source\">Source: [Fine-Grained Segmentation Networks: Self-Supervised Segmentation for Improved Long-Term Visual Localization ](https://arxiv.org/abs/1908.06387)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":402,"papers_with_code":211,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":27,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/visual-localization-on-oxford-radar-robotcar","slug":"visual-localization-on-oxford-radar-robotcar","dataset":"Oxford Radar RobotCar (Full-6)","dataset_url":"/dataset/oxford-radar-robotcar-dataset","rows_in_archive":16,"metrics":["Mean Translation Error"],"first_row_in_archive_order":{"model":"LightLoc","paper_title":"LightLoc: Learning Outdoor LiDAR Localization at Light Speed","paper_url":"/paper/lightloc-learning-outdoor-lidar-localization","paper_date":"2025-03-22","arxiv_id":"2503.17814","code_links":[{"title":"liw95/lightloc","url":"https://github.com/liw95/lightloc"}],"syntology":null}},{"leaderboard":"/sota/visual-localization-on-aachen-day-night-v1-1","slug":"visual-localization-on-aachen-day-night-v1-1","dataset":"Aachen Day-Night v1.1 Benchmark","dataset_url":"/dataset/aachen-day-night-v1-1-benchmark","rows_in_archive":7,"metrics":["Acc@0.25m, 2°","Acc@0.5m, 5°","Acc@5m, 10°"],"first_row_in_archive_order":{"model":"GIM-LoFTR","paper_title":"GIM: Learning Generalizable Image Matcher From Internet Videos","paper_url":"/paper/gim-learning-generalizable-image-matcher-from","paper_date":"2024-02-16","arxiv_id":"2402.11095","code_links":[{"title":"xuelunshen/gim","url":"https://github.com/xuelunshen/gim"}],"syntology":{"n":17,"n_ran":12,"n_unverified":5,"n_pointer_only":0}}},{"leaderboard":"/sota/visual-localization-on-oxford-robotcar-full","slug":"visual-localization-on-oxford-robotcar-full","dataset":"Oxford RobotCar Full","dataset_url":null,"rows_in_archive":6,"metrics":["Mean Translation Error"],"first_row_in_archive_order":{"model":"RobustLoc","paper_title":"RobustLoc: Robust Camera Pose Regression in Challenging Driving Environments","paper_url":"/paper/robustloc-robust-camera-pose-regression-in","paper_date":"2022-11-21","arxiv_id":"2211.11238","code_links":[{"title":"sijieaaa/robustloc","url":"https://github.com/sijieaaa/robustloc"}],"syntology":null}},{"leaderboard":"/sota/visual-localization-on-extended-cmu-seasons","slug":"visual-localization-on-extended-cmu-seasons","dataset":"Extended CMU Seasons","dataset_url":null,"rows_in_archive":1,"metrics":["Acc @ .25m, 2°","Acc @ .5m, 5°","Acc @ 5m, 10°"],"first_row_in_archive_order":{"model":"Patch-NetVLAD","paper_title":"Patch-NetVLAD: Multi-Scale Fusion of Locally-Global Descriptors for Place Recognition","paper_url":"/paper/patch-netvlad-multi-scale-fusion-of-locally","paper_date":"2021-03-02","arxiv_id":"2103.01486","code_links":[{"title":"QVPR/Patch-NetVLAD","url":"https://github.com/QVPR/Patch-NetVLAD"},{"title":"gmberton/VPR-datasets-downloader","url":"https://github.com/gmberton/VPR-datasets-downloader"},{"title":"marialeyvallina/generalized_contrastive_loss","url":"https://github.com/marialeyvallina/generalized_contrastive_loss"},{"title":"taowenyin/PatchNetVLAD","url":"https://github.com/taowenyin/PatchNetVLAD"}],"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/visual-localization-on-robotcar-seasons-v2","slug":"visual-localization-on-robotcar-seasons-v2","dataset":"RobotCar Seasons v2","dataset_url":null,"rows_in_archive":1,"metrics":["Acc @ .25m, 2°","Acc @ .5m, 5°","Acc @ 5m, 10°"],"first_row_in_archive_order":{"model":"Patch-NetVLAD","paper_title":"Patch-NetVLAD: Multi-Scale Fusion of Locally-Global Descriptors for Place Recognition","paper_url":"/paper/patch-netvlad-multi-scale-fusion-of-locally","paper_date":"2021-03-02","arxiv_id":"2103.01486","code_links":[{"title":"QVPR/Patch-NetVLAD","url":"https://github.com/QVPR/Patch-NetVLAD"},{"title":"gmberton/VPR-datasets-downloader","url":"https://github.com/gmberton/VPR-datasets-downloader"},{"title":"marialeyvallina/generalized_contrastive_loss","url":"https://github.com/marialeyvallina/generalized_contrastive_loss"},{"title":"taowenyin/PatchNetVLAD","url":"https://github.com/taowenyin/PatchNetVLAD"}],"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/nclt","name":"NCLT","full_name":"North Campus Long-Term Vision and LiDAR","num_papers_in_archive":124},{"url":"/dataset/cambridge-landmarks","name":"Cambridge Landmarks","full_name":"","num_papers_in_archive":109},{"url":"/dataset/aachen-day-night","name":"Aachen Day-Night","full_name":"","num_papers_in_archive":93},{"url":"/dataset/structured3d","name":"Structured3D","full_name":"","num_papers_in_archive":85},{"url":"/dataset/inloc","name":"InLoc","full_name":"","num_papers_in_archive":67},{"url":"/dataset/oxford-radar-robotcar-dataset","name":"Oxford Radar RobotCar Dataset","full_name":"","num_papers_in_archive":28},{"url":"/dataset/kaist-urban","name":"KAIST Urban","full_name":"","num_papers_in_archive":19},{"url":"/dataset/zind","name":"ZInd","full_name":"Zillow Indoor Dataset","num_papers_in_archive":18},{"url":"/dataset/dublincity","name":"DublinCity","full_name":"","num_papers_in_archive":15},{"url":"/dataset/deeploc","name":"DeepLoc","full_name":"DeepLoc","num_papers_in_archive":8},{"url":"/dataset/aachen-day-night-v1-1-benchmark","name":"Aachen Day-Night v1.1 Benchmark","full_name":"","num_papers_in_archive":7},{"url":"/dataset/indoor-6","name":"Indoor-6","full_name":"","num_papers_in_archive":3},{"url":"/dataset/long-term-visual-localization","name":"Long-term visual localization","full_name":"","num_papers_in_archive":3},{"url":"/dataset/photosynth","name":"PhotoSynth","full_name":null,"num_papers_in_archive":3},{"url":"/dataset/seasondepth","name":"SeasonDepth","full_name":"","num_papers_in_archive":3},{"url":"/dataset/spagbol","name":"SpaGBOL","full_name":"Spatial-Graph-Based Orientated Cross-View Geo-Localisation","num_papers_in_archive":3},{"url":"/dataset/vbr","name":"VBR","full_name":"VBR: A Vision Benchmark in Rome","num_papers_in_archive":3},{"url":"/dataset/virtual-gallery","name":"Virtual Gallery","full_name":"","num_papers_in_archive":2},{"url":"/dataset/crossloc-benchmark-datasets","name":"CrossLoc Benchmark Datasets","full_name":"","num_papers_in_archive":1},{"url":"/dataset/danish-airs-and-grounds","name":"Danish Airs and Grounds","full_name":"","num_papers_in_archive":1},{"url":"/dataset/deeploccross","name":"DeepLocCross","full_name":"DeepLocCross","num_papers_in_archive":1},{"url":"/dataset/fas100k","name":"FAS100K","full_name":null,"num_papers_in_archive":1},{"url":"/dataset/gral","name":"GRAL","full_name":"","num_papers_in_archive":1},{"url":"/dataset/naver-labs-localization-datasets","name":"NAVER LABS Localization Datasets","full_name":"","num_papers_in_archive":1},{"url":"/dataset/peng","name":"PEnG","full_name":"Pose-Enhanced Geo-Localisation","num_papers_in_archive":1},{"url":"/dataset/robust-e-nerf-synthetic-event-dataset","name":"Robust e-NeRF Synthetic Event Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/rover","name":"ROVER","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":211,"tagged_in_all":402,"items":[{"url":"/paper/superglue-learning-feature-matching-with","title":"SuperGlue: Learning Feature Matching with Graph Neural Networks","date":"2019-11-26","arxiv_id":"1911.11763","repositories_listed":19,"syntology":{"n":22,"n_ran":6,"n_unverified":16,"n_pointer_only":6}},{"url":"/paper/pointnetvlad-deep-point-cloud-based-retrieval","title":"PointNetVLAD: Deep Point Cloud Based Retrieval for Large-Scale Place Recognition","date":"2018-04-10","arxiv_id":"1804.03492","repositories_listed":6,"syntology":null},{"url":"/paper/megaloc-one-retrieval-to-place-them-all","title":"MegaLoc: One Retrieval to Place Them All","date":"2025-02-24","arxiv_id":"2502.17237","repositories_listed":4,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/barf-bundle-adjusting-neural-radiance-fields","title":"BARF: Bundle-Adjusting Neural Radiance Fields","date":"2021-04-13","arxiv_id":"2104.06405","repositories_listed":4,"syntology":{"n":19,"n_ran":4,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/loftr-detector-free-local-feature-matching","title":"LoFTR: Detector-Free Local Feature Matching with Transformers","date":"2021-04-01","arxiv_id":"2104.00680","repositories_listed":4,"syntology":{"n":16,"n_ran":14,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/patch-netvlad-multi-scale-fusion-of-locally","title":"Patch-NetVLAD: Multi-Scale Fusion of Locally-Global Descriptors for Place Recognition","date":"2021-03-02","arxiv_id":"2103.01486","repositories_listed":4,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/s2dnet-learning-accurate-correspondences-for","title":"S2DNet: Learning Accurate Correspondences for Sparse-to-Dense Feature Matching","date":"2020-04-03","arxiv_id":"2004.01673","repositories_listed":4,"syntology":null},{"url":"/paper/190503304","title":"Deep Closest Point: Learning Representations for Point Cloud Registration","date":"2019-05-08","arxiv_id":"1905.03304","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/dsac-differentiable-ransac-for-camera","title":"DSAC - Differentiable RANSAC for Camera Localization","date":"2016-11-17","arxiv_id":"1611.05705","repositories_listed":4,"syntology":null},{"url":"/paper/crossloc-scalable-aerial-localization","title":"CrossLoc: Scalable Aerial Localization Assisted by Multimodal Synthetic Data","date":"2021-12-16","arxiv_id":"2112.09081","repositories_listed":3,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/s2dnet-learning-image-features-for-accurate","title":"S2DNet: Learning Image Features for Accurate Sparse-to-Dense Matching","date":"2020-08-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/adalam-revisiting-handcrafted-outlier","title":"AdaLAM: Revisiting Handcrafted Outlier Detection","date":"2020-06-07","arxiv_id":"2006.04250","repositories_listed":3,"syntology":{"n":19,"n_ran":0,"n_unverified":19,"n_pointer_only":0}},{"url":"/paper/university-1652-a-multi-view-multi-source","title":"University-1652: A Multi-view Multi-source Benchmark for Drone-based Geo-localization","date":"2020-02-27","arxiv_id":"2002.12186","repositories_listed":3,"syntology":{"n":15,"n_ran":1,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/neural-guided-ransac-learning-where-to-sample","title":"Neural-Guided RANSAC: Learning Where to Sample Model Hypotheses","date":"2019-05-10","arxiv_id":"1905.04132","repositories_listed":3,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/from-coarse-to-fine-robust-hierarchical","title":"From Coarse to Fine: Robust Hierarchical Localization at Large Scale","date":"2018-12-09","arxiv_id":"1812.03506","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":2}},{"url":"/paper/visual-localization-under-appearance-change-a","title":"Visual Localization Under Appearance Change: Filtering Approaches","date":"2018-11-20","arxiv_id":"1811.08063","repositories_listed":3,"syntology":null},{"url":"/paper/neighbourhood-consensus-networks","title":"Neighbourhood Consensus Networks","date":"2018-10-24","arxiv_id":"1810.10510","repositories_listed":3,"syntology":{"n":21,"n_ran":5,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","arxiv_id":"2308.12966","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/lightglue-local-feature-matching-at-light","title":"LightGlue: Local Feature Matching at Light Speed","date":"2023-06-23","arxiv_id":"2306.13643","repositories_listed":2,"syntology":{"n":40,"n_ran":18,"n_unverified":22,"n_pointer_only":0}},{"url":"/paper/orienternet-visual-localization-in-2d-public","title":"OrienterNet: Visual Localization in 2D Public Maps with Neural Matching","date":"2023-04-04","arxiv_id":"2304.02009","repositories_listed":2,"syntology":null},{"url":"/paper/gluestick-robust-image-matching-by-sticking","title":"GlueStick: Robust Image Matching by Sticking Points and Lines Together","date":"2023-04-04","arxiv_id":"2304.02008","repositories_listed":2,"syntology":{"n":29,"n_ran":20,"n_unverified":9,"n_pointer_only":15}},{"url":"/paper/hypliloc-towards-effective-lidar-pose","title":"HypLiLoc: Towards Effective LiDAR Pose Regression with Hyperbolic Fusion","date":"2023-04-03","arxiv_id":"2304.00932","repositories_listed":2,"syntology":null},{"url":"/paper/piccolo-point-cloud-centric-omnidirectional","title":"PICCOLO: Point Cloud-Centric Omnidirectional Localization","date":"2021-08-14","arxiv_id":"2108.06545","repositories_listed":2,"syntology":null},{"url":"/paper/learning-multi-scene-absolute-pose-regression","title":"Learning Multi-Scene Absolute Pose Regression with Transformers","date":"2021-03-21","arxiv_id":"2103.11468","repositories_listed":2,"syntology":{"n":20,"n_ran":10,"n_unverified":10,"n_pointer_only":20}},{"url":"/paper/kimera-from-slam-to-spatial-perception-with","title":"Kimera: from SLAM to Spatial Perception with 3D Dynamic Scene Graphs","date":"2021-01-18","arxiv_id":"2101.06894","repositories_listed":2,"syntology":null},{"url":"/paper/vr-caps-a-virtual-environment-for-capsule","title":"VR-Caps: A Virtual Environment for Capsule Endoscopy","date":"2020-08-29","arxiv_id":"2008.12949","repositories_listed":2,"syntology":null},{"url":"/paper/robust-image-retrieval-based-visual","title":"Robust Image Retrieval-based Visual Localization using Kapture","date":"2020-07-27","arxiv_id":"2007.13867","repositories_listed":2,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/cmrnet-map-and-camera-agnostic-monocular","title":"CMRNet++: Map and Camera Agnostic Monocular Visual Localization in LiDAR Maps","date":"2020-04-20","arxiv_id":"2004.13795","repositories_listed":2,"syntology":null},{"url":"/paper/dynamic-slam-semantic-monocular-visual","title":"Dynamic-SLAM: Semantic monocular visual localization and mapping based on deep learning in dynamic environment","date":"2019-07-06","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/panoramic-annular-localizer-tackling-the","title":"Panoramic Annular Localizer: Tackling the Variation Challenges of Outdoor Localization Using Panoramic Annular Images and Active Deep Descriptors","date":"2019-05-14","arxiv_id":"1905.05425","repositories_listed":2,"syntology":null}],"syntology_records":17,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}