{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/spatial-reasoning/papers/2","list_of":"/task/spatial-reasoning","task":"Spatial Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":453,"counts":{"archive_papers_tagged":453,"with_a_code_link":198,"where_syntology_ran_a_sample":70,"not_listed_spam_title":0,"listed":453,"listed_where_code_ran":70,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":59,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":59,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/spatial-reasoning","prev":"/task/spatial-reasoning","next":"/task/spatial-reasoning/papers/3","papers":[{"url":"/paper/thinking-in-space-how-multimodal-large","slug":"thinking-in-space-how-multimodal-large","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","date":"2024-12-18","arxiv_id":"2412.14171","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thinking-in-space-how-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2412.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14171"}},"official":{"repos":["vision-x-nyu/thinking-in-space"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphere-a-hierarchical-evaluation-on-spatial","slug":"sphere-a-hierarchical-evaluation-on-spatial","title":"SPHERE: A Hierarchical Evaluation on Spatial Perception and Reasoning for Vision-Language Models","date":"2024-12-17","arxiv_id":"2412.12693","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphere-a-hierarchical-evaluation-on-spatial#ran","syntology_url":"https://syntology.ai/paper/2412.12693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12693"}},"official":{"repos":["zwenyu/SPHERE-VLM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emma-x-an-embodied-multimodal-action-model","slug":"emma-x-an-embodied-multimodal-action-model","title":"Emma-X: An Embodied Multimodal Action Model with Grounded Chain of Thought and Look-ahead Spatial Reasoning","date":"2024-12-16","arxiv_id":"2412.11974","repositories_listed":1,"syntology":null},{"url":"/paper/sat-spatial-aptitude-training-for-multimodal","slug":"sat-spatial-aptitude-training-for-multimodal","title":"SAT: Dynamic Spatial Aptitude Training for Multimodal Language Models","date":"2024-12-10","arxiv_id":"2412.07755","repositories_listed":1,"syntology":null},{"url":"/paper/taco-learning-multi-modal-action-models-with","slug":"taco-learning-multi-modal-action-models-with","title":"TACO: Learning Multi-modal Action Models with Synthetic Chains-of-Thought-and-Action","date":"2024-12-07","arxiv_id":"2412.05479","repositories_listed":1,"syntology":null},{"url":"/paper/can-large-language-models-reason-about-the","slug":"can-large-language-models-reason-about-the","title":"Can Large Language Models Reason about the Region Connection Calculus?","date":"2024-11-29","arxiv_id":"2411.19589","repositories_listed":1,"syntology":null},{"url":"/paper/grid-augumented-vision-a-simple-yet-effective","slug":"grid-augumented-vision-a-simple-yet-effective","title":"Grid-augmented vision: A simple yet effective approach for enhanced spatial understanding in multi-modal agents","date":"2024-11-27","arxiv_id":"2411.18270","repositories_listed":1,"syntology":null},{"url":"/paper/apt-architectural-planning-and-text-to","slug":"apt-architectural-planning-and-text-to","title":"APT: Architectural Planning and Text-to-Blueprint Construction Using Large Language Models for Open-World Agents","date":"2024-11-26","arxiv_id":"2411.17255","repositories_listed":1,"syntology":null},{"url":"/paper/citywalker-learning-embodied-urban-navigation","slug":"citywalker-learning-embodied-urban-navigation","title":"CityWalker: Learning Embodied Urban Navigation from Web-Scale Videos","date":"2024-11-26","arxiv_id":"2411.17820","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citywalker-learning-embodied-urban-navigation#ran","syntology_url":"https://syntology.ai/paper/2411.17820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17820"}},"official":{"repos":["ai4ce/CityWalker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/drivemllm-a-benchmark-for-spatial","slug":"drivemllm-a-benchmark-for-spatial","title":"DriveMLLM: A Benchmark for Spatial Understanding with Multimodal Large Language Models in Autonomous Driving","date":"2024-11-20","arxiv_id":"2411.13112","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drivemllm-a-benchmark-for-spatial#ran","syntology_url":"https://syntology.ai/paper/2411.13112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13112"}},"official":{"repos":["xiandaguo/drive-mllm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-analysis-on-spatial-reasoning","slug":"an-empirical-analysis-on-spatial-reasoning","title":"An Empirical Analysis on Spatial Reasoning Capabilities of Large Multimodal Models","date":"2024-11-09","arxiv_id":"2411.06048","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-navigation-with-vision-language","slug":"end-to-end-navigation-with-vision-language","title":"End-to-End Navigation with Vision Language Models: Transforming Spatial Reasoning into Question-Answering","date":"2024-11-08","arxiv_id":"2411.05755","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-navigation-with-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.05755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05755"}},"official":{"repos":["Jirl-upenn/VLMnav"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rocket-1-master-open-world-interaction-with","slug":"rocket-1-master-open-world-interaction-with","title":"ROCKET-1: Mastering Open-World Interaction with Visual-Temporal Context Prompting","date":"2024-10-23","arxiv_id":"2410.17856","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rocket-1-master-open-world-interaction-with#ran","syntology_url":"https://syntology.ai/paper/2410.17856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17856"}},"official":{"repos":["CraftJarvis/ROCKET-1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-vision-language-models-represent-space-and","slug":"do-vision-language-models-represent-space-and","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","date":"2024-10-22","arxiv_id":"2410.17385","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/do-vision-language-models-represent-space-and#ran","syntology_url":"https://syntology.ai/paper/2410.17385","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17385"}},"official":{"repos":["sled-group/COMFORT"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/locality-alignment-improves-vision-language","slug":"locality-alignment-improves-vision-language","title":"Locality Alignment Improves Vision-Language Models","date":"2024-10-14","arxiv_id":"2410.11087","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/locality-alignment-improves-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.11087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11087"}},"official":null}},{"url":"/paper/ing-vp-mllms-cannot-play-easy-vision-based","slug":"ing-vp-mllms-cannot-play-easy-vision-based","title":"ING-VP: MLLMs cannot Play Easy Vision-based Games Yet","date":"2024-10-09","arxiv_id":"2410.06555","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-code-llms-on-geospatial-code","slug":"evaluation-of-code-llms-on-geospatial-code","title":"Evaluation of Code LLMs on Geospatial Code Generation","date":"2024-10-06","arxiv_id":"2410.04617","repositories_listed":1,"syntology":null},{"url":"/paper/polymath-a-challenging-multi-modal","slug":"polymath-a-challenging-multi-modal","title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","date":"2024-10-06","arxiv_id":"2410.14702","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polymath-a-challenging-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2410.14702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14702"}},"official":{"repos":["polymathbenchmark/PolyMATH"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openkd-opening-prompt-diversity-for-zero-and","slug":"openkd-opening-prompt-diversity-for-zero-and","title":"OpenKD: Opening Prompt Diversity for Zero- and Few-shot Keypoint Detection","date":"2024-09-30","arxiv_id":"2409.19899","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/openkd-opening-prompt-diversity-for-zero-and#ran","syntology_url":"https://syntology.ai/paper/2409.19899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19899"}},"official":{"repos":["alanlusun/openkd"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videoinsta-zero-shot-long-video-understanding","slug":"videoinsta-zero-shot-long-video-understanding","title":"VideoINSTA: Zero-shot Long Video Understanding via Informative Spatial-Temporal Reasoning with LLMs","date":"2024-09-30","arxiv_id":"2409.20365","repositories_listed":1,"syntology":null},{"url":"/paper/can-vision-language-models-learn-from-visual","slug":"can-vision-language-models-learn-from-visual","title":"Can Vision Language Models Learn from Visual Demonstrations of Ambiguous Spatial Reasoning?","date":"2024-09-25","arxiv_id":"2409.17080","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-logical-reasoning-in-large-language-1","slug":"enhancing-logical-reasoning-in-large-language-1","title":"Enhancing Logical Reasoning in Large Language Models through Graph-based Synthetic Data","date":"2024-09-19","arxiv_id":"2409.12437","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-temporal-spatial-reasoning","slug":"unleashing-the-temporal-spatial-reasoning","title":"Unleashing the Temporal-Spatial Reasoning Capacity of GPT for Training-Free Audio and Language Referenced Video Object Segmentation","date":"2024-08-28","arxiv_id":"2408.15876","repositories_listed":1,"syntology":null},{"url":"/paper/narrowing-the-gap-between-vision-and-action","slug":"narrowing-the-gap-between-vision-and-action","title":"Narrowing the Gap between Vision and Action in Navigation","date":"2024-08-19","arxiv_id":"2408.10388","repositories_listed":1,"syntology":null},{"url":"/paper/show-don-t-tell-evaluating-large-language","slug":"show-don-t-tell-evaluating-large-language","title":"Show, Don't Tell: Evaluating Large Language Models Beyond Textual Understanding with ChildPlay","date":"2024-07-12","arxiv_id":"2407.11068","repositories_listed":1,"syntology":null},{"url":"/paper/learning-action-and-reasoning-centric-image","slug":"learning-action-and-reasoning-centric-image","title":"Learning Action and Reasoning-Centric Image Editing from Videos and Simulations","date":"2024-07-03","arxiv_id":"2407.03471","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-action-and-reasoning-centric-image#ran","syntology_url":"https://syntology.ai/paper/2407.03471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03471"}},"official":{"repos":["McGill-NLP/AURORA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/is-a-picture-worth-a-thousand-words-delving","slug":"is-a-picture-worth-a-thousand-words-delving","title":"Is A Picture Worth A Thousand Words? Delving Into Spatial Reasoning for Vision Language Models","date":"2024-06-21","arxiv_id":"2406.14852","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":5,"n_instrument":7,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":7,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/is-a-picture-worth-a-thousand-words-delving#ran","syntology_url":"https://syntology.ai/paper/2406.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14852"}},"official":{"repos":["jiayuww/SpatialEval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/citygpt-empowering-urban-spatial-cognition-of","slug":"citygpt-empowering-urban-spatial-cognition-of","title":"CityGPT: Empowering Urban Spatial Cognition of Large Language Models","date":"2024-06-20","arxiv_id":"2406.13948","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citygpt-empowering-urban-spatial-cognition-of#ran","syntology_url":"https://syntology.ai/paper/2406.13948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13948"}},"official":{"repos":["tsinghua-fib-lab/citygpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neuro-symbolic-training-for-reasoning-over","slug":"neuro-symbolic-training-for-reasoning-over","title":"Neuro-symbolic Training for Reasoning over Spatial Language","date":"2024-06-19","arxiv_id":"2406.13828","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/neuro-symbolic-training-for-reasoning-over#ran","syntology_url":"https://syntology.ai/paper/2406.13828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13828"}},"official":{"repos":["premsrit/SPARTUNQChain"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/spatialbot-precise-spatial-understanding-with","slug":"spatialbot-precise-spatial-understanding-with","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","date":"2024-06-19","arxiv_id":"2406.13642","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spatialbot-precise-spatial-understanding-with#ran","syntology_url":"https://syntology.ai/paper/2406.13642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13642"}},"official":{"repos":["baai-dcai/spatialbot"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-sketchpad-sketching-as-a-visual-chain","slug":"visual-sketchpad-sketching-as-a-visual-chain","title":"Visual Sketchpad: Sketching as a Visual Chain of Thought for Multimodal Language Models","date":"2024-06-13","arxiv_id":"2406.09403","repositories_listed":1,"syntology":null},{"url":"/paper/flow-of-reasoning-efficient-training-of-llm","slug":"flow-of-reasoning-efficient-training-of-llm","title":"Flow of Reasoning:Training LLMs for Divergent Problem Solving with Minimal Examples","date":"2024-06-09","arxiv_id":"2406.05673","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-of-reasoning-efficient-training-of-llm#ran","syntology_url":"https://syntology.ai/paper/2406.05673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05673"}},"official":{"repos":["yu-fangxu/for"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sparc-and-sparp-spatial-reasoning","slug":"sparc-and-sparp-spatial-reasoning","title":"SpaRC and SpaRP: Spatial Reasoning Characterization and Path Generation for Understanding Spatial Reasoning Capability of Large Language Models","date":"2024-06-07","arxiv_id":"2406.04566","repositories_listed":1,"syntology":null},{"url":"/paper/topviewrs-vision-language-models-as-top-view","slug":"topviewrs-vision-language-models-as-top-view","title":"TopViewRS: Vision-Language Models as Top-View Spatial Reasoners","date":"2024-06-04","arxiv_id":"2406.02537","repositories_listed":1,"syntology":null},{"url":"/paper/spatialrgpt-grounded-spatial-reasoning-in","slug":"spatialrgpt-grounded-spatial-reasoning-in","title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Models","date":"2024-06-03","arxiv_id":"2406.01584","repositories_listed":1,"syntology":null},{"url":"/paper/when-llms-step-into-the-3d-world-a-survey-and","slug":"when-llms-step-into-the-3d-world-a-survey-and","title":"When LLMs step into the 3D World: A Survey and Meta-Analysis of 3D Tasks via Multi-modal Large Language Models","date":"2024-05-16","arxiv_id":"2405.10255","repositories_listed":1,"syntology":null},{"url":"/paper/dara-domain-and-relation-aware-adapters-make","slug":"dara-domain-and-relation-aware-adapters-make","title":"DARA: Domain- and Relation-aware Adapters Make Parameter-efficient Tuning for Visual Grounding","date":"2024-05-10","arxiv_id":"2405.06217","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-localize-objects-improves-spatial","slug":"learning-to-localize-objects-improves-spatial","title":"Learning to Localize Objects Improves Spatial Reasoning in Visual-LLMs","date":"2024-04-11","arxiv_id":"2404.07449","repositories_listed":1,"syntology":null},{"url":"/paper/visualization-of-thought-elicits-spatial","slug":"visualization-of-thought-elicits-spatial","title":"Mind's Eye of LLMs: Visualization-of-Thought Elicits Spatial Reasoning in Large Language Models","date":"2024-04-04","arxiv_id":"2404.03622","repositories_listed":1,"syntology":null},{"url":"/paper/getting-it-right-improving-spatial","slug":"getting-it-right-improving-spatial","title":"Getting it Right: Improving Spatial Consistency in Text-to-Image Models","date":"2024-04-01","arxiv_id":"2404.01197","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/getting-it-right-improving-spatial#ran","syntology_url":"https://syntology.ai/paper/2404.01197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01197"}},"official":{"repos":["SPRIGHT-T2I/SPRIGHT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-spatial-relations-in-text-only","slug":"grounding-spatial-relations-in-text-only","title":"Grounding Spatial Relations in Text-Only Language Models","date":"2024-03-20","arxiv_id":"2403.13666","repositories_listed":1,"syntology":null},{"url":"/paper/llmarena-assessing-capabilities-of-large","slug":"llmarena-assessing-capabilities-of-large","title":"LLMArena: Assessing Capabilities of Large Language Models in Dynamic Multi-Agent Environments","date":"2024-02-26","arxiv_id":"2402.16499","repositories_listed":1,"syntology":null},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/advancing-spatial-reasoning-in-large-language","slug":"advancing-spatial-reasoning-in-large-language","title":"Advancing Spatial Reasoning in Large Language Models: An In-Depth Evaluation and Enhancement Using the StepGame Benchmark","date":"2024-01-08","arxiv_id":"2401.03991","repositories_listed":1,"syntology":null},{"url":"/paper/location-aware-modular-biencoder-for-tourism","slug":"location-aware-modular-biencoder-for-tourism","title":"Location Aware Modular Biencoder for Tourism Question Answering","date":"2024-01-04","arxiv_id":"2401.02187","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-spatio-temporal-decoupling-for","slug":"hierarchical-spatio-temporal-decoupling-for","title":"Hierarchical Spatio-temporal Decoupling for Text-to-Video Generation","date":"2023-12-07","arxiv_id":"2312.04483","repositories_listed":1,"syntology":null},{"url":"/paper/inherent-limitations-of-llms-regarding","slug":"inherent-limitations-of-llms-regarding","title":"Inherent limitations of LLMs regarding spatial information","date":"2023-12-05","arxiv_id":"2312.03042","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-road-with-gpt-4v-ision-early","slug":"on-the-road-with-gpt-4v-ision-early","title":"On the Road with GPT-4V(ision): Early Explorations of Visual-Language Model on Autonomous Driving","date":"2023-11-09","arxiv_id":"2311.05332","repositories_listed":1,"syntology":null},{"url":"/paper/disentangling-extraction-and-reasoning-in","slug":"disentangling-extraction-and-reasoning-in","title":"Disentangling Extraction and Reasoning in Multi-hop Spatial Reasoning","date":"2023-10-25","arxiv_id":"2310.16731","repositories_listed":1,"syntology":null},{"url":"/paper/depwignn-a-depth-wise-graph-neural-network","slug":"depwignn-a-depth-wise-graph-neural-network","title":"DepWiGNN: A Depth-wise Graph Neural Network for Multi-hop Spatial Reasoning in Text","date":"2023-10-19","arxiv_id":"2310.12557","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-are-zero-shot-reward","slug":"vision-language-models-are-zero-shot-reward","title":"Vision-Language Models are Zero-Shot Reward Models for Reinforcement Learning","date":"2023-10-19","arxiv_id":"2310.12921","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vision-language-models-are-zero-shot-reward#ran","syntology_url":"https://syntology.ai/paper/2310.12921","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12921"}},"official":{"repos":["alignmentresearch/vlmrm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-large-language-models-be-good-path","slug":"can-large-language-models-be-good-path","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","date":"2023-10-05","arxiv_id":"2310.03249","repositories_listed":1,"syntology":{"n":28,"n_ran":28,"n_constructed":0,"n_ran_checked":28,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":28,"n_pointer_only":28,"phrase":"28 ran (of which 0 constructed an object rather than computing a result; 28 with no instrument failure: 0 honoured, 0 violated, 28 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-large-language-models-be-good-path#ran","syntology_url":"https://syntology.ai/paper/2310.03249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03249"}},"official":{"repos":["mohamedaghzal/llms-as-path-planners"],"state":"official (archive's flag): 28 ran","n_ran":28,"n_constructed":0,"n_ran_no_instrument_failure":28,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/talk2bev-language-enhanced-bird-s-eye-view","slug":"talk2bev-language-enhanced-bird-s-eye-view","title":"Talk2BEV: Language-enhanced Bird's-eye View Maps for Autonomous Driving","date":"2023-10-03","arxiv_id":"2310.02251","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/talk2bev-language-enhanced-bird-s-eye-view#ran","syntology_url":"https://syntology.ai/paper/2310.02251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02251"}},"official":null}},{"url":"/paper/smartplay-a-benchmark-for-llms-as-intelligent","slug":"smartplay-a-benchmark-for-llms-as-intelligent","title":"SmartPlay: A Benchmark for LLMs as Intelligent Agents","date":"2023-10-02","arxiv_id":"2310.01557","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/smartplay-a-benchmark-for-llms-as-intelligent#ran","syntology_url":"https://syntology.ai/paper/2310.01557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01557"}},"official":{"repos":["microsoft/smartplay"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dense-2d-3d-indoor-prediction-with-sound-via","slug":"dense-2d-3d-indoor-prediction-with-sound-via","title":"Dense 2D-3D Indoor Prediction with Sound via Aligned Cross-Modal Distillation","date":"2023-09-20","arxiv_id":"2309.11081","repositories_listed":1,"syntology":null},{"url":"/paper/stupd-a-synthetic-dataset-for-spatial-and","slug":"stupd-a-synthetic-dataset-for-spatial-and","title":"STUPD: A Synthetic Dataset for Spatial and Temporal Relation Reasoning","date":"2023-09-13","arxiv_id":"2309.06680","repositories_listed":1,"syntology":null},{"url":"/paper/droppos-pre-training-vision-transformers-by-1","slug":"droppos-pre-training-vision-transformers-by-1","title":"DropPos: Pre-Training Vision Transformers by Reconstructing Dropped Positions","date":"2023-09-07","arxiv_id":"2309.03576","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/droppos-pre-training-vision-transformers-by-1#ran","syntology_url":"https://syntology.ai/paper/2309.03576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03576"}},"official":{"repos":["haochen-wang409/droppos"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/spacenli-evaluating-the-consistency-of","slug":"spacenli-evaluating-the-consistency-of","title":"SpaceNLI: Evaluating the Consistency of Predicting Inferences in Space","date":"2023-07-05","arxiv_id":"2307.02269","repositories_listed":1,"syntology":null},{"url":"/paper/a-universal-semantic-geometric-representation","slug":"a-universal-semantic-geometric-representation","title":"A Universal Semantic-Geometric Representation for Robotic Manipulation","date":"2023-06-18","arxiv_id":"2306.10474","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-universal-semantic-geometric-representation#ran","syntology_url":"https://syntology.ai/paper/2306.10474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10474"}},"official":{"repos":["TongZhangTHU/sgr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-task-synthesis-for-visual-programming","slug":"neural-task-synthesis-for-visual-programming","title":"Neural Task Synthesis for Visual Programming","date":"2023-05-26","arxiv_id":"2305.18342","repositories_listed":1,"syntology":null},{"url":"/paper/egohumans-an-egocentric-3d-multi-human","slug":"egohumans-an-egocentric-3d-multi-human","title":"EgoHumans: An Egocentric 3D Multi-Human Benchmark","date":"2023-05-25","arxiv_id":"2305.16487","repositories_listed":1,"syntology":null},{"url":"/paper/are-llms-the-master-of-all-trades-exploring","slug":"are-llms-the-master-of-all-trades-exploring","title":"Are LLMs the Master of All Trades? : Exploring Domain-Agnostic Reasoning Skills of LLMs","date":"2023-03-22","arxiv_id":"2303.12810","repositories_listed":1,"syntology":null},{"url":"/paper/conceptfusion-open-set-multimodal-3d-mapping","slug":"conceptfusion-open-set-multimodal-3d-mapping","title":"ConceptFusion: Open-set Multimodal 3D Mapping","date":"2023-02-14","arxiv_id":"2302.07241","repositories_listed":1,"syntology":null},{"url":"/paper/translating-natural-language-to-planning","slug":"translating-natural-language-to-planning","title":"Translating Natural Language to Planning Goals with Large-Language Models","date":"2023-02-10","arxiv_id":"2302.05128","repositories_listed":1,"syntology":null},{"url":"/paper/are-deep-neural-networks-smarter-than-second","slug":"are-deep-neural-networks-smarter-than-second","title":"Are Deep Neural Networks SMARTer than Second Graders?","date":"2022-12-20","arxiv_id":"2212.09993","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-deep-neural-networks-smarter-than-second#ran","syntology_url":"https://syntology.ai/paper/2212.09993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09993"}},"official":{"repos":["merlresearch/SMART"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/location-aware-self-supervised-transformers","slug":"location-aware-self-supervised-transformers","title":"Location-Aware Self-Supervised Transformers for Semantic Segmentation","date":"2022-12-05","arxiv_id":"2212.02400","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/location-aware-self-supervised-transformers#ran","syntology_url":"https://syntology.ai/paper/2212.02400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02400"}},"official":{"repos":["google-research/scenic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lovis-learning-orientation-and-visual-signals","slug":"lovis-learning-orientation-and-visual-signals","title":"LOViS: Learning Orientation and Visual Signals for Vision and Language Navigation","date":"2022-09-26","arxiv_id":"2209.12723","repositories_listed":1,"syntology":null},{"url":"/paper/knowing-earlier-what-right-means-to-you-a","slug":"knowing-earlier-what-right-means-to-you-a","title":"Knowing Earlier what Right Means to You: A Comprehensive VQA Dataset for Grounding Relative Directions via Multi-Task Learning","date":"2022-07-06","arxiv_id":"2207.02624","repositories_listed":1,"syntology":null},{"url":"/paper/translating-place-related-questions-to","slug":"translating-place-related-questions-to","title":"Translating Place-Related Questions to GeoSPARQL Queries","date":"2022-05-06","arxiv_id":"2205.03067","repositories_listed":1,"syntology":null},{"url":"/paper/explicit-object-relation-alignment-for-vision-1","slug":"explicit-object-relation-alignment-for-vision-1","title":"Explicit Object Relation Alignment for Vision and Language Navigation","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/stepgame-a-new-benchmark-for-robust-multi-hop-1","slug":"stepgame-a-new-benchmark-for-robust-multi-hop-1","title":"StepGame: A New Benchmark for Robust Multi-Hop Spatial Reasoning in Texts","date":"2022-04-18","arxiv_id":"2204.08292","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stepgame-a-new-benchmark-for-robust-multi-hop-1#ran","syntology_url":"https://syntology.ai/paper/2204.08292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08292"}},"official":{"repos":["ZhengxiangShi/StepGame"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/capturing-shape-information-with-multi-scale","slug":"capturing-shape-information-with-multi-scale","title":"Capturing Shape Information with Multi-Scale Topological Loss Terms for 3D Reconstruction","date":"2022-03-03","arxiv_id":"2203.01703","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/capturing-shape-information-with-multi-scale#ran","syntology_url":"https://syntology.ai/paper/2203.01703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01703"}},"official":{"repos":["marrlab/shapr_torch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deepssn-a-deep-convolutional-neural-network","slug":"deepssn-a-deep-convolutional-neural-network","title":"DeepSSN: a deep convolutional neural network to assess spatial scene similarity","date":"2022-02-07","arxiv_id":"2202.04755","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-spatio-temporal-layouts-for","slug":"revisiting-spatio-temporal-layouts-for","title":"Revisiting spatio-temporal layouts for compositional action recognition","date":"2021-11-02","arxiv_id":"2111.01936","repositories_listed":1,"syntology":null},{"url":"/paper/indonli-a-natural-language-inference-dataset","slug":"indonli-a-natural-language-inference-dataset","title":"IndoNLI: A Natural Language Inference Dataset for Indonesian","date":"2021-10-27","arxiv_id":"2110.14566","repositories_listed":1,"syntology":null},{"url":"/paper/cliport-what-and-where-pathways-for-robotic","slug":"cliport-what-and-where-pathways-for-robotic","title":"CLIPort: What and Where Pathways for Robotic Manipulation","date":"2021-09-24","arxiv_id":"2109.12098","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-natural-language-instructions-can","slug":"grounding-natural-language-instructions-can","title":"Grounding Natural Language Instructions: Can Large Language Models Capture Spatial Information?","date":"2021-09-17","arxiv_id":"2109.08634","repositories_listed":1,"syntology":null},{"url":"/paper/sornet-spatial-object-centric-representations","slug":"sornet-spatial-object-centric-representations","title":"SORNet: Spatial Object-Centric Representations for Sequential Manipulation","date":"2021-09-08","arxiv_id":"2109.03891","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sornet-spatial-object-centric-representations#ran","syntology_url":"https://syntology.ai/paper/2109.03891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03891"}},"official":{"repos":["wentaoyuan/sornet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sbevnet-end-to-end-deep-stereo-layout-1","slug":"sbevnet-end-to-end-deep-stereo-layout-1","title":"SBEVNet: End-to-End Deep Stereo Layout Estimation","date":"2021-05-25","arxiv_id":"2105.11705","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-spatial-reasoning-on-multi-view","slug":"contrastive-spatial-reasoning-on-multi-view","title":"Self-supervised Spatial Reasoning on Multi-View Line Drawings","date":"2021-04-27","arxiv_id":"2104.13433","repositories_listed":1,"syntology":null},{"url":"/paper/spartqa-a-textual-question-answering","slug":"spartqa-a-textual-question-answering","title":"SpartQA: : A Textual Question Answering Benchmark for Spatial Reasoning","date":"2021-04-12","arxiv_id":"2104.05832","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/spartqa-a-textual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2104.05832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.05832"}},"official":{"repos":["HLR/SpartQA_generation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-scale-gcn-assisted-two-stage-network","slug":"multi-scale-gcn-assisted-two-stage-network","title":"Multi-scale GCN-assisted two-stage network for joint segmentation of retinal layers and disc in peripapillary OCT images","date":"2021-02-09","arxiv_id":"2102.04799","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-consistency-distilling-spatial","slug":"grounding-consistency-distilling-spatial","title":"Grounding Consistency: Distilling Spatial Common Sense for Precise Visual Relationship Detection","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/guided-navigation-from-multiple-viewpoints","slug":"guided-navigation-from-multiple-viewpoints","title":"Guided Navigation from Multiple Viewpoints using Qualitative Spatial Reasoning","date":"2020-11-03","arxiv_id":"2011.01397","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-language-spatial-relations-to-2d","slug":"decoding-language-spatial-relations-to-2d","title":"Decoding Language Spatial Relations to 2D Spatial Arrangements","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bist-bi-directional-spatio-temporal-reasoning","slug":"bist-bi-directional-spatio-temporal-reasoning","title":"BiST: Bi-directional Spatio-Temporal Reasoning for Video-Grounded Dialogues","date":"2020-10-20","arxiv_id":"2010.10095","repositories_listed":1,"syntology":null},{"url":"/paper/joint-spatio-textual-reasoning-for-answering","slug":"joint-spatio-textual-reasoning-for-answering","title":"Joint Spatio-Textual Reasoning for Answering Tourism Questions","date":"2020-09-28","arxiv_id":"2009.13613","repositories_listed":1,"syntology":null},{"url":"/paper/spatially-aware-multimodal-transformers-for","slug":"spatially-aware-multimodal-transformers-for","title":"Spatially Aware Multimodal Transformers for TextVQA","date":"2020-07-23","arxiv_id":"2007.12146","repositories_listed":1,"syntology":null},{"url":"/paper/se-kge-a-location-aware-knowledge-graph","slug":"se-kge-a-location-aware-knowledge-graph","title":"SE-KGE: A Location-Aware Knowledge Graph Embedding Model for Geographic Question Answering and Spatial Semantic Lifting","date":"2020-04-25","arxiv_id":"2004.14171","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/se-kge-a-location-aware-knowledge-graph#ran","syntology_url":"https://syntology.ai/paper/2004.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14171"}},"official":null}},{"url":"/paper/spare3d-a-dataset-for-spatial-reasoning-on","slug":"spare3d-a-dataset-for-spatial-reasoning-on","title":"SPARE3D: A Dataset for SPAtial REasoning on Three-View Line Drawings","date":"2020-03-31","arxiv_id":"2003.14034","repositories_listed":1,"syntology":null},{"url":"/paper/pix2shape-towards-unsupervised-learning-of-3d","slug":"pix2shape-towards-unsupervised-learning-of-3d","title":"Pix2Shape: Towards Unsupervised Learning of 3D Scenes from Images using a View-based Representation","date":"2020-03-23","arxiv_id":"2003.14166","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/pix2shape-towards-unsupervised-learning-of-3d#ran","syntology_url":"https://syntology.ai/paper/2003.14166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.14166"}},"official":{"repos":["rajeswar18/pix2shape"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/spatialsense-an-adversarially-crowdsourced","slug":"spatialsense-an-adversarially-crowdsourced","title":"SpatialSense: An Adversarially Crowdsourced Benchmark for Spatial Relation Recognition","date":"2019-08-07","arxiv_id":"1908.02660","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/spatialsense-an-adversarially-crowdsourced#ran","syntology_url":"https://syntology.ai/paper/1908.02660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.02660"}},"official":{"repos":["princeton-vl/SpatialSense"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/encoding-spatial-relations-from-natural","slug":"encoding-spatial-relations-from-natural","title":"Encoding Spatial Relations from Natural Language","date":"2018-07-04","arxiv_id":"1807.01670","repositories_listed":1,"syntology":null},{"url":"/paper/cilantro-a-lean-versatile-and-efficient","slug":"cilantro-a-lean-versatile-and-efficient","title":"cilantro: A Lean, Versatile, and Efficient Library for Point Cloud Data Processing","date":"2018-07-01","arxiv_id":"1807.00399","repositories_listed":1,"syntology":null},{"url":"/paper/a-trajectory-calculus-for-qualitative-spatial","slug":"a-trajectory-calculus-for-qualitative-spatial","title":"A Trajectory Calculus for Qualitative Spatial Reasoning Using Answer Set Programming","date":"2018-04-19","arxiv_id":"1804.07088","repositories_listed":1,"syntology":null},{"url":"/paper/representation-learning-for-grounded-spatial","slug":"representation-learning-for-grounded-spatial","title":"Representation Learning for Grounded Spatial Reasoning","date":"2017-07-13","arxiv_id":"1707.03938","repositories_listed":1,"syntology":null},{"url":null,"slug":"mindjourney-test-time-scaling-with-world","title":"MindJourney: Test-Time Scaling with World Models for Spatial Reasoning","date":"2025-07-16","arxiv_id":"2507.12508","repositories_listed":0,"syntology":null},{"url":null,"slug":"embrace-3k-embodied-reasoning-and-action-in","title":"EmbRACE-3K: Embodied Reasoning and Action in Complex Environments","date":"2025-07-14","arxiv_id":"2507.10548","repositories_listed":0,"syntology":null}],"record_sha256":"b41504f185a68d314abba2a52a22ca69bd03ac39093fdf128acfdc150fd166af","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}