{"url":"/task/spatial-reasoning","name":"Spatial Reasoning","slug":"spatial-reasoning","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":453,"papers_with_code":198,"benchmarks":2,"benchmark_tables_in_archive":16,"benchmark_tables_shown":16,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/spatial-reasoning-on-6-dof-spatialbench","slug":"spatial-reasoning-on-6-dof-spatialbench","dataset":"6-DoF SpatialBench","dataset_url":null,"rows_in_archive":7,"metrics":["Total","Position-rel","Position-abs","Orientation-rel","Orientation-abs"],"first_row_in_archive_order":{"model":"SoFar","paper_title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","paper_url":"/paper/sofar-language-grounded-orientation-bridges","paper_date":"2025-02-18","arxiv_id":"2502.13143","code_links":[{"title":"qizekun/SoFar","url":"https://github.com/qizekun/SoFar"},{"title":"zhangwenyao1/open6dor_v2_execution","url":"https://github.com/zhangwenyao1/open6dor_v2_execution"}],"syntology":null}},{"leaderboard":"/sota/spatial-reasoning-on-embspatial-bench","slug":"spatial-reasoning-on-embspatial-bench","dataset":"EmbSpatial-Bench","dataset_url":"/dataset/embspatial-bench","rows_in_archive":5,"metrics":["Generation"],"first_row_in_archive_order":{"model":"SoFar","paper_title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","paper_url":"/paper/sofar-language-grounded-orientation-bridges","paper_date":"2025-02-18","arxiv_id":"2502.13143","code_links":[{"title":"qizekun/SoFar","url":"https://github.com/qizekun/SoFar"},{"title":"zhangwenyao1/open6dor_v2_execution","url":"https://github.com/zhangwenyao1/open6dor_v2_execution"}],"syntology":null}},{"leaderboard":null,"slug":"spatial-reasoning-on-3dsrbench","dataset":"3DSRBench","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-blink","dataset":"BlINK","dataset_url":"/dataset/blink","rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-cvbench","dataset":"cvbench","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-mmiu","dataset":"MMIU","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-mmvp","dataset":"MMVP","dataset_url":"/dataset/musicqa-dataset","rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-q-spatial-bench","dataset":"Q-Spatial-Bench","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-qspatialbench-plus","dataset":"QSpatialBench-Plus","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-qspatialbench-scannet","dataset":"QSpatialBench-ScanNet","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-realworldqa","dataset":"RealWorldQA","dataset_url":"/dataset/realworldqa","rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-spatialbench","dataset":"SpatialBench","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-spatialsense","dataset":"SpatialSense","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-vgbench","dataset":"VGBench","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-vsi-bench-8","dataset":"VSI-Bench_8","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null},{"leaderboard":null,"slug":"spatial-reasoning-on-vsr-zeroshot","dataset":"VSR-ZeroShot","dataset_url":null,"rows_in_archive":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null}],"datasets":[{"url":"/dataset/blink","name":"BlINK","full_name":"","num_papers_in_archive":58},{"url":"/dataset/musicqa-dataset","name":"MMVP","full_name":"","num_papers_in_archive":53},{"url":"/dataset/embspatial-bench","name":"EmbSpatial-Bench","full_name":"EmbSpatial-Bench: Benchmarking Spatial Understanding for Embodied Tasks with Large Vision-Language Models","num_papers_in_archive":6},{"url":"/dataset/repair","name":"RePAIR Dataset","full_name":"","num_papers_in_archive":2},{"url":"/dataset/realworldqa","name":"RealWorldQA","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[{"url":"/task/visual-question-answering-1","name":"Visual Question Answering"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":198,"tagged_in_all":453,"items":[{"url":"/paper/spatial-memory-for-context-reasoning-in","title":"Spatial Memory for Context Reasoning in Object Detection","date":"2017-04-13","arxiv_id":"1704.04224","repositories_listed":35,"syntology":null},{"url":"/paper/visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_unverified":35,"n_pointer_only":0}},{"url":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","arxiv_id":"2310.03744","repositories_listed":9,"syntology":{"n":9,"n_ran":6,"n_unverified":3,"n_pointer_only":8}},{"url":"/paper/minigpt-4-enhancing-vision-language","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","date":"2023-04-20","arxiv_id":"2304.10592","repositories_listed":6,"syntology":null},{"url":"/paper/long-range-arena-a-benchmark-for-efficient-1","title":"Long Range Arena: A Benchmark for Efficient Transformers","date":"2020-11-08","arxiv_id":"2011.04006","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/visual-spatial-reasoning","title":"Visual Spatial Reasoning","date":"2022-04-30","arxiv_id":"2205.00363","repositories_listed":4,"syntology":{"n":10,"n_ran":5,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/touchdown-natural-language-navigation-and","title":"Touchdown: Natural Language Navigation and Spatial Reasoning in Visual Street Environments","date":"2018-11-29","arxiv_id":"1811.12354","repositories_listed":4,"syntology":null},{"url":"/paper/guesswhat-visual-object-discovery-through","title":"GuessWhat?! Visual object discovery through multi-modal dialogue","date":"2016-11-23","arxiv_id":"1611.08481","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/mapeval-a-map-based-evaluation-of-geo-spatial","title":"MapEval: A Map-Based Evaluation of Geo-Spatial Reasoning in Foundation Models","date":"2024-12-31","arxiv_id":"2501.00316","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/probing-the-limitations-of-multimodal","title":"Probing the limitations of multimodal language models for chemistry and materials research","date":"2024-11-25","arxiv_id":"2411.16955","repositories_listed":3,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/no-blind-spots-full-surround-multi-object","title":"No Blind Spots: Full-Surround Multi-Object Tracking for Autonomous Vehicles using Cameras & LiDARs","date":"2018-02-23","arxiv_id":"1802.08755","repositories_listed":3,"syntology":null},{"url":"/paper/infigui-r1-advancing-multimodal-gui-agents","title":"InfiGUI-R1: Advancing Multimodal GUI Agents from Reactive Actors to Deliberative Reasoners","date":"2025-04-19","arxiv_id":"2504.14239","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/spatial-r1-enhancing-mllms-in-video-spatial","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","date":"2025-04-02","arxiv_id":"2504.01805","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/sonata-self-supervised-learning-of-reliable","title":"Sonata: Self-Supervised Learning of Reliable Point Representations","date":"2025-03-20","arxiv_id":"2503.16429","repositories_listed":2,"syntology":{"n":19,"n_ran":5,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/iref-vla-a-benchmark-for-interactive","title":"IRef-VLA: A Benchmark for Interactive Referential Grounding with Imperfect Language in 3D Scenes","date":"2025-03-20","arxiv_id":"2503.17406","repositories_listed":2,"syntology":null},{"url":"/paper/sofar-language-grounded-orientation-bridges","title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","date":"2025-02-18","arxiv_id":"2502.13143","repositories_listed":2,"syntology":null},{"url":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied","title":"CityEQA: A Hierarchical LLM Agent on Embodied Question Answering Benchmark in City Space","date":"2025-02-18","arxiv_id":"2502.12532","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":4}},{"url":"/paper/on-the-planning-abilities-of-openai-s-o1","title":"On The Planning Abilities of OpenAI's o1 Models: Feasibility, Optimality, and Generalizability","date":"2024-09-30","arxiv_id":"2409.19924","repositories_listed":2,"syntology":null},{"url":"/paper/reframing-spatial-reasoning-evaluation-in","title":"Reframing Spatial Reasoning Evaluation in Language Models: A Real-World Simulation Benchmark for Qualitative Reasoning","date":"2024-05-23","arxiv_id":"2405.15064","repositories_listed":2,"syntology":null},{"url":"/paper/seeing-the-roads-through-the-trees-a","title":"Seeing the roads through the trees: A benchmark for modeling spatial dependencies with aerial imagery","date":"2024-01-12","arxiv_id":"2401.06762","repositories_listed":2,"syntology":null},{"url":"/paper/what-s-up-with-vision-language-models","title":"What's \"up\" with vision-language models? Investigating their struggle with spatial reasoning","date":"2023-10-30","arxiv_id":"2310.19785","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":2}},{"url":"/paper/qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","arxiv_id":"2308.12966","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/chat-3d-data-efficiently-tuning-large","title":"Chat-3D: Data-efficiently Tuning Large Language Model for Universal Dialogue of 3D Scenes","date":"2023-08-17","arxiv_id":"2308.08769","repositories_listed":2,"syntology":null},{"url":"/paper/act3d-infinite-resolution-action-detection","title":"Act3D: 3D Feature Field Transformers for Multi-Task Robotic Manipulation","date":"2023-06-30","arxiv_id":"2306.17817","repositories_listed":2,"syntology":null},{"url":"/paper/llm-grounded-diffusion-enhancing-prompt","title":"LLM-grounded Diffusion: Enhancing Prompt Understanding of Text-to-Image Diffusion Models with Large Language Models","date":"2023-05-23","arxiv_id":"2305.13655","repositories_listed":2,"syntology":null},{"url":"/paper/reclip-a-strong-zero-shot-baseline-for-1","title":"ReCLIP: A Strong Zero-Shot Baseline for Referring Expression Comprehension","date":"2022-04-12","arxiv_id":"2204.05991","repositories_listed":2,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/teaching-agents-how-to-map-spatial-reasoning","title":"Teaching Agents how to Map: Spatial Reasoning for Multi-Object Navigation","date":"2021-07-13","arxiv_id":"2107.06011","repositories_listed":2,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/spartqa-a-textual-question-answering-1","title":"SPARTQA: A Textual Question Answering Benchmark for Spatial Reasoning","date":"2021-06-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-egospheric-spatial-memory","title":"End-to-End Egospheric Spatial Memory","date":"2021-02-15","arxiv_id":"2102.07764","repositories_listed":2,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}