{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/spatial-reasoning/papers/4","list_of":"/task/spatial-reasoning","task":"Spatial Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":453,"counts":{"archive_papers_tagged":453,"with_a_code_link":198,"where_syntology_ran_a_sample":70,"not_listed_spam_title":0,"listed":453,"listed_where_code_ran":70,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":59,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":59,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/spatial-reasoning","prev":"/task/spatial-reasoning/papers/3","next":"/task/spatial-reasoning/papers/5","papers":[{"url":null,"slug":"exploring-spatial-language-grounding-through","title":"Exploring Spatial Language Grounding Through Referring Expressions","date":"2025-02-04","arxiv_id":"2502.04359","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-rag-spatial-retrieval-augmented","title":"Spatial-RAG: Spatial Retrieval Augmented Generation for Real-World Geospatial Reasoning Questions","date":"2025-02-04","arxiv_id":"2502.18470","repositories_listed":0,"syntology":null},{"url":null,"slug":"vl-nav-real-time-vision-language-navigation","title":"VL-Nav: Real-time Vision-Language Navigation with Spatial Reasoning","date":"2025-02-02","arxiv_id":"2502.00931","repositories_listed":0,"syntology":null},{"url":null,"slug":"rls3-rl-based-synthetic-sample-selection-to","title":"RLS3: RL-Based Synthetic Sample Selection to Enhance Spatial Reasoning in Vision-Language Models for Indoor Autonomous Perception","date":"2025-01-31","arxiv_id":"2501.18880","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-moe-a-mixture-of-experts-multi-modal-llm","title":"3D-MoE: A Mixture-of-Experts Multi-modal LLM for 3D Vision and Pose Diffusion via Rectified Flow","date":"2025-01-28","arxiv_id":"2501.16698","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-visualization-and-optimization","title":"Bridging Visualization and Optimization: Multimodal Large Language Models on Graph-Structured Combinatorial Optimization","date":"2025-01-21","arxiv_id":"2501.11968","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatialcot-advancing-spatial-reasoning","title":"SpatialCoT: Advancing Spatial Reasoning through Coordinate Alignment and Chain-of-Thought for Embodied Task Planning","date":"2025-01-17","arxiv_id":"2501.10074","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-scene-understanding-for-vision","title":"Embodied Scene Understanding for Vision Language Models via MetaVQA","date":"2025-01-15","arxiv_id":"2501.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxdepthnet-real-time-monocular-3d-object","title":"AuxDepthNet: Real-Time Monocular 3D Object Detection with Depth-Sensitive Features","date":"2025-01-07","arxiv_id":"2501.03700","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-semantics-programming-in-3d-gaussian","title":"Chain of Semantics Programming in 3D Gaussian Splatting Representation for 3D Vision Grounding","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"r2c-mapping-room-to-chessboard-to-unlock-llm","title":"R2C: Mapping Room to Chessboard to Unlock LLM As Low-Level Action Planner","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ske-layout-spatial-knowledge-enhanced-layout","title":"SKE-Layout: Spatial Knowledge Enhanced Layout Generation with LLMs","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial457-a-diagnostic-benchmark-for-6d","title":"Spatial457: A Diagnostic Benchmark for 6D Spatial Reasoning of Large Mutimodal Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spatialclip-learning-3d-aware-image","title":"SpatialCLIP: Learning 3D-aware Image Representations from Spatially Discriminative Language","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cad-gpt-synthesising-cad-construction","title":"CAD-GPT: Synthesising CAD Construction Sequence with Spatial Reasoning-Enhanced Multimodal LLMs","date":"2024-12-27","arxiv_id":"2412.19663","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-of-thoughts-extracting-and-following","title":"Path-of-Thoughts: Extracting and Following Paths for Robust Relational Reasoning with Large Language Models","date":"2024-12-23","arxiv_id":"2412.17963","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-multimodal-language-models-really","title":"Do Multimodal Language Models Really Understand Direction? A Benchmark for Compass Direction Reasoning","date":"2024-12-21","arxiv_id":"2412.16599","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathematical-definition-and-systematization","title":"Mathematical Definition and Systematization of Puzzle Rules","date":"2024-12-18","arxiv_id":"2501.01433","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-contrastive-framework","title":"A dual contrastive framework","date":"2024-12-13","arxiv_id":"2412.10348","repositories_listed":0,"syntology":null},{"url":null,"slug":"geo-llava-a-large-multi-modal-model-for","title":"Geo-LLaVA: A Large Multi-Modal Model for Solving Geometry Math Problems with Meta In-Context Learning","date":"2024-12-12","arxiv_id":"2412.10455","repositories_listed":0,"syntology":null},{"url":null,"slug":"visionarena-230k-real-world-user-vlm","title":"VisionArena: 230K Real World User-VLM Conversations with Preference Labels","date":"2024-12-11","arxiv_id":"2412.08687","repositories_listed":0,"syntology":null},{"url":null,"slug":"3dsrbench-a-comprehensive-3d-spatial","title":"3DSRBench: A Comprehensive 3D Spatial Reasoning Benchmark","date":"2024-12-10","arxiv_id":"2412.07825","repositories_listed":0,"syntology":null},{"url":null,"slug":"videosavi-self-aligned-video-language-models","title":"VideoSAVi: Self-Aligned Video Language Models without Human Supervision","date":"2024-12-01","arxiv_id":"2412.00624","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pipeline-of-neural-symbolic-integration-to","title":"Dspy-based Neural-Symbolic Pipeline to Enhance Spatial Reasoning in LLMs","date":"2024-11-27","arxiv_id":"2411.18564","repositories_listed":0,"syntology":null},{"url":null,"slug":"robospatial-teaching-spatial-understanding-to","title":"RoboSpatial: Teaching Spatial Understanding to 2D and 3D Vision-Language Models for Robotics","date":"2024-11-25","arxiv_id":"2411.16537","repositories_listed":0,"syntology":null},{"url":null,"slug":"topv-nav-unlocking-the-top-view-spatial","title":"TopV-Nav: Unlocking the Top-View Spatial Reasoning Potential of MLLM for Zero-shot Object Navigation","date":"2024-11-25","arxiv_id":"2411.16425","repositories_listed":0,"syntology":null},{"url":null,"slug":"balrog-benchmarking-agentic-llm-and-vlm","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","date":"2024-11-20","arxiv_id":"2411.13543","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-linguistic-agent-towards-collaborative","title":"Visual-Linguistic Agent: Towards Collaborative Contextual Object Reasoning","date":"2024-11-15","arxiv_id":"2411.10252","repositories_listed":0,"syntology":null},{"url":null,"slug":"architect-generating-vivid-and-interactive-3d","title":"Architect: Generating Vivid and Interactive 3D Scenes with Hierarchical 2D Inpainting","date":"2024-11-14","arxiv_id":"2411.09823","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-s-spatial-intelligence-evaluating-ai-s","title":"AI's Spatial Intelligence: Evaluating AI's Understanding of Spatial Transformations in PSVT:R and Augmented Reality","date":"2024-11-09","arxiv_id":"2411.06269","repositories_listed":0,"syntology":null},{"url":"/paper/gpt-4o-system-card","slug":"gpt-4o-system-card","title":"GPT-4o System Card","date":"2024-10-25","arxiv_id":"2410.21276","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-feature-enhanced-knowledge-graph","title":"Geometric Feature Enhanced Knowledge Graph Embedding and Spatial Reasoning","date":"2024-10-24","arxiv_id":"2410.18345","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-am-i-and-what-will-i-see-an-auto","title":"Where Am I and What Will I See: An Auto-Regressive Model for Spatial Localization and View Prediction","date":"2024-10-24","arxiv_id":"2410.18962","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparkle-mastering-basic-spatial-capabilities","title":"Sparkle: Mastering Basic Spatial Capabilities in Vision Language Models Elicits Generalization to Composite Spatial Reasoning","date":"2024-10-21","arxiv_id":"2410.16162","repositories_listed":0,"syntology":null},{"url":null,"slug":"aerial-vision-and-language-navigation-via","title":"Aerial Vision-and-Language Navigation via Semantic-Topo-Metric Representation Guided LLM Reasoning","date":"2024-10-11","arxiv_id":"2410.08500","repositories_listed":0,"syntology":null},{"url":null,"slug":"testing-gpt-4-o1-preview-on-math-and-science","title":"Testing GPT-4-o1-preview on math and science problems: A follow-up study","date":"2024-10-11","arxiv_id":"2410.22340","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-spatial-reasoning-with-open","title":"Structured Spatial Reasoning with Open Vocabulary Object Detectors","date":"2024-10-09","arxiv_id":"2410.07394","repositories_listed":0,"syntology":null},{"url":null,"slug":"spartun3d-situated-spatial-understanding-of","title":"SPARTUN3D: Situated Spatial Understanding of 3D World in Large Language Models","date":"2024-10-04","arxiv_id":"2410.03878","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-conjuring-multi-user-runtime","title":"Social Conjuring: Multi-User Runtime Collaboration with AI in Building Virtual 3D Worlds","date":"2024-09-30","arxiv_id":"2410.00274","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-reasoning-and-planning-for-deep","title":"Spatial Reasoning and Planning for Deep Embodied Agents","date":"2024-09-28","arxiv_id":"2409.19479","repositories_listed":0,"syntology":null},{"url":null,"slug":"dare-diverse-visual-question-answering-with","title":"DARE: Diverse Visual Question Answering with Robustness Evaluation","date":"2024-09-26","arxiv_id":"2409.18023","repositories_listed":0,"syntology":null},{"url":null,"slug":"tag-map-a-text-based-map-for-spatial","title":"Tag Map: A Text-Based Map for Spatial Reasoning and Navigation with Large Language Models","date":"2024-09-23","arxiv_id":"2409.15451","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-paths-with-reference-objects-elicit","title":"Reasoning Paths with Reference Objects Elicit Quantitative Spatial Reasoning in Large Vision-Language Models","date":"2024-09-15","arxiv_id":"2409.09788","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionflow-equivariant-accurate-and-efficient","title":"ActionFlow: Equivariant, Accurate, and Efficient Policies with Spatially Symmetric Flow Matching","date":"2024-09-06","arxiv_id":"2409.04576","repositories_listed":0,"syntology":null},{"url":null,"slug":"cog-ga-a-large-language-models-based","title":"Cog-GA: A Large Language Models-based Generative Agent for Vision-Language Navigation in Continuous Environments","date":"2024-09-04","arxiv_id":"2409.02522","repositories_listed":0,"syntology":null},{"url":null,"slug":"aeroverse-uav-agent-benchmark-suite-for","title":"AeroVerse: UAV-Agent Benchmark Suite for Simulating, Pre-training, Finetuning, and Evaluating Aerospace Embodied World Models","date":"2024-08-28","arxiv_id":"2408.15511","repositories_listed":0,"syntology":null},{"url":null,"slug":"atari-gpt-investigating-the-capabilities-of","title":"Atari-GPT: Benchmarking Multimodal Large Language Models as Low-Level Policies in Atari Games","date":"2024-08-28","arxiv_id":"2408.15950","repositories_listed":0,"syntology":null},{"url":null,"slug":"poly2vec-polymorphic-encoding-of-geospatial","title":"Poly2Vec: Polymorphic Fourier-Based Encoding of Geospatial Objects for GeoAI Applications","date":"2024-08-27","arxiv_id":"2408.14806","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llm-be-a-good-path-planner-based-on","title":"Can LLM be a Good Path Planner based on Prompt Engineering? Mitigating the Hallucination for Path Planning","date":"2024-08-23","arxiv_id":"2408.13184","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-hype-a-dispassionate-look-at","title":"Beyond the Hype: A dispassionate look at vision-language models in medical scenario","date":"2024-08-16","arxiv_id":"2408.08704","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenegpt-a-language-model-for-3d-scene","title":"SceneGPT: A Language Model for 3D Scene Understanding","date":"2024-08-13","arxiv_id":"2408.06926","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-a-foundation-model-for-space-based","title":"Space-LLaVA: a Vision-Language Model Adapted to Extraterrestrial Applications","date":"2024-08-12","arxiv_id":"2408.05924","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02231","title":"REVISION: Rendering Tools Enable Spatial Fidelity in Vision-Language Models","date":"2024-08-05","arxiv_id":"2408.02231","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00754","title":"Coarse Correspondences Boost Spatial-Temporal Reasoning in Multimodal Language Model","date":"2024-08-01","arxiv_id":"2408.00754","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-know-about-up-enhancing-spatial-reasoning","title":"I Know About \"Up\"! Enhancing Spatial Reasoning in Visual Language Models Through 3D Reconstruction","date":"2024-07-19","arxiv_id":"2407.14133","repositories_listed":0,"syntology":null},{"url":null,"slug":"opensu3d-open-world-3d-scene-understanding","title":"OpenSU3D: Open World 3D Scene Understanding using Foundation Models","date":"2024-07-19","arxiv_id":"2407.14279","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-llm-benchmark-based-on-the-minecraft","title":"A LLM Benchmark based on the Minecraft Builder Dialog Agent Task","date":"2024-07-17","arxiv_id":"2407.12734","repositories_listed":0,"syntology":null},{"url":null,"slug":"grasp-a-grid-based-benchmark-for-evaluating","title":"GRASP: A Grid-Based Benchmark for Evaluating Commonsense Spatial Reasoning","date":"2024-07-02","arxiv_id":"2407.01892","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowvqa-mapping-multimodal-logic-in-visual","title":"FlowVQA: Mapping Multimodal Logic in Visual Question Answering with Flowcharts","date":"2024-06-27","arxiv_id":"2406.19237","repositories_listed":0,"syntology":null},{"url":null,"slug":"whiteboard-of-thought-thinking-step-by-step","title":"Whiteboard-of-Thought: Thinking Step-by-Step Across Modalities","date":"2024-06-20","arxiv_id":"2406.14562","repositories_listed":0,"syntology":null},{"url":null,"slug":"gsr-bench-a-benchmark-for-grounded-spatial","title":"GSR-BENCH: A Benchmark for Grounded Spatial Reasoning Evaluation via Multimodal LLMs","date":"2024-06-19","arxiv_id":"2406.13246","repositories_listed":0,"syntology":null},{"url":null,"slug":"wildvision-evaluating-vision-language-models","title":"WildVision: Evaluating Vision-Language Models in the Wild with Human Preferences","date":"2024-06-16","arxiv_id":"2406.11069","repositories_listed":0,"syntology":null},{"url":"/paper/robopoint-a-vision-language-model-for-spatial","slug":"robopoint-a-vision-language-model-for-spatial","title":"RoboPoint: A Vision-Language Model for Spatial Affordance Prediction for Robotics","date":"2024-06-15","arxiv_id":"2406.10721","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-geospatial-in-the-common-crawl","title":"Quantifying Geospatial in the Common Crawl Corpus","date":"2024-06-07","arxiv_id":"2406.04952","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-large-language-models-create-new","title":"Can Large Language Models Create New Knowledge for Spatial Reasoning Tasks?","date":"2024-05-23","arxiv_id":"2405.14379","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-human-motion-in-3d-scenes-from","title":"Generating Human Motion in 3D Scenes from Text Descriptions","date":"2024-05-13","arxiv_id":"2405.07784","repositories_listed":0,"syntology":null},{"url":null,"slug":"robohop-segment-based-topological-map","title":"RoboHop: Segment-based Topological Map Representation for Open-World Visual Navigation","date":"2024-05-09","arxiv_id":"2405.05792","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-human-vision-the-role-of-large-vision","title":"Beyond Human Vision: The Role of Large Vision Language Models in Microscope Image Analysis","date":"2024-05-01","arxiv_id":"2405.00876","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-thinking-inverse-graphics-with-large","title":"Re-Thinking Inverse Graphics With Large Language Models","date":"2024-04-23","arxiv_id":"2404.15228","repositories_listed":0,"syntology":null},{"url":null,"slug":"hammr-hierarchical-multimodal-react-agents","title":"HAMMR: HierArchical MultiModal React agents for generic VQA","date":"2024-04-08","arxiv_id":"2404.05465","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-faced-by-large-language-models-in","title":"Challenges Faced by Large Language Models in Solving Multi-Agent Flocking","date":"2024-04-06","arxiv_id":"2404.04752","repositories_listed":0,"syntology":null},{"url":null,"slug":"see-imagine-plan-discovering-and","title":"SpatialPIN: Enhancing Spatial Reasoning Capabilities of Vision-Language Models through Prompting and Interacting 3D Priors","date":"2024-03-18","arxiv_id":"2403.13438","repositories_listed":0,"syntology":null},{"url":null,"slug":"jstr-joint-spatio-temporal-reasoning-for","title":"JSTR: Joint Spatio-Temporal Reasoning for Event-based Moving Object Detection","date":"2024-03-12","arxiv_id":"2403.07436","repositories_listed":0,"syntology":null},{"url":null,"slug":"divcon-divide-and-conquer-for-progressive","title":"DivCon: Divide and Conquer for Progressive Text-to-Image Generation","date":"2024-03-11","arxiv_id":"2403.06400","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-region-guidance-improving","title":"Contrastive Region Guidance: Improving Grounding in Vision-Language Models without Training","date":"2024-03-04","arxiv_id":"2403.02325","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-surprising-failure-multimodal-llms-and-the","title":"A Surprising Failure? Multimodal LLMs and the NLVR Challenge","date":"2024-02-26","arxiv_id":"2402.17793","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivevlm-the-convergence-of-autonomous","title":"DriveVLM: The Convergence of Autonomous Driving and Large Vision-Language Models","date":"2024-02-19","arxiv_id":"2402.12289","repositories_listed":0,"syntology":null},{"url":null,"slug":"pivot-iterative-visual-prompting-elicits","title":"PIVOT: Iterative Visual Prompting Elicits Actionable Knowledge for VLMs","date":"2024-02-12","arxiv_id":"2402.07872","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-for-foundation-models-in-autonomous","title":"A Survey for Foundation Models in Autonomous Driving","date":"2024-02-02","arxiv_id":"2402.01105","repositories_listed":0,"syntology":null},{"url":"/paper/spatialvlm-endowing-vision-language-models","slug":"spatialvlm-endowing-vision-language-models","title":"SpatialVLM: Endowing Vision-Language Models with Spatial Reasoning Capabilities","date":"2024-01-22","arxiv_id":"2401.12168","repositories_listed":0,"syntology":null},{"url":null,"slug":"starcraftimage-a-dataset-for-prototyping-1","title":"StarCraftImage: A Dataset For Prototyping Spatial Reasoning Methods For Multi-Agent Environments","date":"2024-01-09","arxiv_id":"2401.04290","repositories_listed":0,"syntology":null},{"url":null,"slug":"distortions-in-judged-spatial-relations-in","title":"Distortions in Judged Spatial Relations in Large Language Models","date":"2024-01-08","arxiv_id":"2401.04218","repositories_listed":0,"syntology":null},{"url":null,"slug":"lidar-llm-exploring-the-potential-of-large","title":"LiDAR-LLM: Exploring the Potential of Large Language Models for 3D LiDAR Understanding","date":"2023-12-21","arxiv_id":"2312.14074","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-and-improving-the-spatial-reasoning","title":"Exploring and Improving the Spatial Reasoning Abilities of Large Language Models","date":"2023-12-02","arxiv_id":"2312.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"followeval-a-multi-dimensional-benchmark-for","title":"FollowEval: A Multi-Dimensional Benchmark for Assessing the Instruction-Following Capability of Large Language Models","date":"2023-11-16","arxiv_id":"2311.09829","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-visual","title":"Evaluating Robustness of Visual Representations for Object Assembly Task Requiring Spatio-Geometrical Reasoning","date":"2023-10-15","arxiv_id":"2310.09943","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-symbolic-reasoning-into-neural","title":"Integrating Symbolic Reasoning into Neural Generative Models for Design Generation","date":"2023-10-13","arxiv_id":"2310.09383","repositories_listed":0,"syntology":null},{"url":null,"slug":"slotgnn-unsupervised-discovery-of-multi","title":"SlotGNN: Unsupervised Discovery of Multi-Object Representations and Visual Dynamics","date":"2023-10-06","arxiv_id":"2310.04617","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evaluation-of-chatgpt-4-s-qualitative","title":"An Evaluation of ChatGPT-4's Qualitative Spatial Reasoning Capabilities in RCC-8","date":"2023-09-27","arxiv_id":"2309.15577","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-camera-bird-s-eye-view-perception-for","title":"Multi-camera Bird's Eye View Perception for Autonomous Driving","date":"2023-09-16","arxiv_id":"2309.09080","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-grounded-visual-spatial-reasoning-in","title":"Towards Grounded Visual Spatial Reasoning in Multi-Modal Vision Language Models","date":"2023-08-18","arxiv_id":"2308.09778","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-goal-navigation-with-recursive","title":"Object Goal Navigation with Recursive Implicit Maps","date":"2023-08-10","arxiv_id":"2308.05602","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-intelligence-of-a-self-driving-car","title":"Spatial Intelligence of a Self-driving Car and Rule-Based Decision Making","date":"2023-08-02","arxiv_id":"2308.01085","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-text-to-image-generation-with","title":"Controllable Text-to-Image Generation with GPT-4","date":"2023-05-29","arxiv_id":"2305.18583","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-algorithms-for-allen-s-interval","title":"Improved Algorithms for Allen's Interval Algebra by Dynamic Programming with Sublinear Partitioning","date":"2023-05-25","arxiv_id":"2305.15950","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-patches-to-objects-exploiting-spatial","title":"From Patches to Objects: Exploiting Spatial Reasoning for Better Visual Representations","date":"2023-05-21","arxiv_id":"2305.12384","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-reasoning-for-scene-generation","title":"Contextual Reasoning for Scene Generation (Technical Report)","date":"2023-05-03","arxiv_id":"2305.02255","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialectical-language-model-evaluation-an","title":"Dialectical language model evaluation: An initial appraisal of the commonsense spatial reasoning abilities of LLMs","date":"2023-04-22","arxiv_id":"2304.11164","repositories_listed":0,"syntology":null},{"url":null,"slug":"morpho-logic-from-a-topos-perspective","title":"Morpho-logic from a Topos Perspective: Application to symbolic AI","date":"2023-03-08","arxiv_id":"2303.04895","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperdimensional-computing-with-spiking","title":"Hyperdimensional Computing with Spiking-Phasor Neurons","date":"2023-02-28","arxiv_id":"2303.00066","repositories_listed":0,"syntology":null}],"record_sha256":"75629e492e3de01b19bd25d1ef964c785e21281ee36659211903351f48862fa3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}