{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/spatial-reasoning/papers/ran/1","list_of":"/task/spatial-reasoning","task":"Spatial Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,70],"of":70,"counts":{"archive_papers_tagged":453,"with_a_code_link":198,"where_syntology_ran_a_sample":70,"not_listed_spam_title":0,"listed":453,"listed_where_code_ran":70,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":59,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":59,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/spatial-reasoning/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/scaling-rl-to-long-videos","slug":"scaling-rl-to-long-videos","title":"Scaling RL to Long Videos","date":"2025-07-10","arxiv_id":"2507.07966","repositories_listed":1,"syntology":{"n":24,"n_ran":18,"n_constructed":1,"n_ran_checked":14,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":13,"n_pointer_only":1,"phrase":"18 ran (of which 1 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/scaling-rl-to-long-videos#ran","syntology_url":"https://syntology.ai/paper/2507.07966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.07966"}},"official":{"repos":["hiyouga/easyr1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/learning-from-videos-for-3d-world-enhancing","slug":"learning-from-videos-for-3d-world-enhancing","title":"Learning from Videos for 3D World: Enhancing MLLMs with 3D Vision Geometry Priors","date":"2025-05-30","arxiv_id":"2505.24625","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-from-videos-for-3d-world-enhancing#ran","syntology_url":"https://syntology.ai/paper/2505.24625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24625"}},"official":null}},{"url":"/paper/uni-mumer-unified-multi-task-fine-tuning-of","slug":"uni-mumer-unified-multi-task-fine-tuning-of","title":"Uni-MuMER: Unified Multi-Task Fine-Tuning of Vision-Language Model for Handwritten Mathematical Expression Recognition","date":"2025-05-29","arxiv_id":"2505.23566","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":1,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-mumer-unified-multi-task-fine-tuning-of#ran","syntology_url":"https://syntology.ai/paper/2505.23566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23566"}},"official":{"repos":["bflameswift/uni-mumer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thinkgeo-evaluating-tool-augmented-agents-for","slug":"thinkgeo-evaluating-tool-augmented-agents-for","title":"ThinkGeo: Evaluating Tool-Augmented Agents for Remote Sensing Tasks","date":"2025-05-29","arxiv_id":"2505.23752","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/thinkgeo-evaluating-tool-augmented-agents-for#ran","syntology_url":"https://syntology.ai/paper/2505.23752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23752"}},"official":{"repos":["mbzuai-oryx/thinkgeo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-action-model-with-open-world","slug":"vision-language-action-model-with-open-world","title":"ChatVLA-2: Vision-Language-Action Model with Open-World Embodied Reasoning from Pretrained Knowledge","date":"2025-05-28","arxiv_id":"2505.21906","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vision-language-action-model-with-open-world#ran","syntology_url":"https://syntology.ai/paper/2505.21906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21906"}},"official":null}},{"url":"/paper/mineanybuild-benchmarking-spatial-planning","slug":"mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","arxiv_id":"2505.20148","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineanybuild-benchmarking-spatial-planning#ran","syntology_url":"https://syntology.ai/paper/2505.20148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20148"}},"official":{"repos":["mineanybuild/mineanybuild"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/spatialscore-towards-unified-evaluation-for","slug":"spatialscore-towards-unified-evaluation-for","title":"SpatialScore: Towards Unified Evaluation for Multimodal Spatial Understanding","date":"2025-05-22","arxiv_id":"2505.17012","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatialscore-towards-unified-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.17012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17012"}},"official":{"repos":["haoningwu3639/SpatialScore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/got-r1-unleashing-reasoning-capability-of","slug":"got-r1-unleashing-reasoning-capability-of","title":"GoT-R1: Unleashing Reasoning Capability of MLLM for Visual Generation with Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.17022","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":6,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/got-r1-unleashing-reasoning-capability-of#ran","syntology_url":"https://syntology.ai/paper/2505.17022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17022"}},"official":{"repos":["gogoduan/got-r1"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/from-seeing-to-doing-bridging-reasoning-and","slug":"from-seeing-to-doing-bridging-reasoning-and","title":"From Seeing to Doing: Bridging Reasoning and Decision for Robotic Manipulation","date":"2025-05-13","arxiv_id":"2505.08548","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/from-seeing-to-doing-bridging-reasoning-and#ran","syntology_url":"https://syntology.ai/paper/2505.08548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.08548"}},"official":{"repos":["pickxiguapi/embodied-fsd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-visual-chain-of-thought","slug":"unsupervised-visual-chain-of-thought","title":"Unsupervised Visual Chain-of-Thought Reasoning via Preference Optimization","date":"2025-04-25","arxiv_id":"2504.18397","repositories_listed":1,"syntology":{"n":30,"n_ran":25,"n_constructed":3,"n_ran_checked":15,"n_instrument":10,"n_unverified":5,"n_honours":1,"n_violates":3,"n_no_contract":11,"n_pointer_only":30,"phrase":"25 ran (of which 3 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 3 violated, 11 with no contract checked; 10 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unsupervised-visual-chain-of-thought#ran","syntology_url":"https://syntology.ai/paper/2504.18397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18397"}},"official":{"repos":["kesenzhao/uv-cot"],"state":"official (archive's flag): 24 ran","n_ran":24,"n_constructed":3,"n_ran_no_instrument_failure":14,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/infigui-r1-advancing-multimodal-gui-agents","slug":"infigui-r1-advancing-multimodal-gui-agents","title":"InfiGUI-R1: Advancing Multimodal GUI Agents from Reactive Actors to Deliberative Reasoners","date":"2025-04-19","arxiv_id":"2504.14239","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/infigui-r1-advancing-multimodal-gui-agents#ran","syntology_url":"https://syntology.ai/paper/2504.14239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14239"}},"official":{"repos":["reallm-labs/infigui-r1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spatial-r1-enhancing-mllms-in-video-spatial","slug":"spatial-r1-enhancing-mllms-in-video-spatial","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","date":"2025-04-02","arxiv_id":"2504.01805","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatial-r1-enhancing-mllms-in-video-spatial#ran","syntology_url":"https://syntology.ai/paper/2504.01805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01805"}},"official":{"repos":["ouyangkun10/spacer","ouyangkun10/spatial-r1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-scaling-for-complex-tasks","slug":"inference-time-scaling-for-complex-tasks","title":"Inference-Time Scaling for Complex Tasks: Where We Stand and What Lies Ahead","date":"2025-03-31","arxiv_id":"2504.00294","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inference-time-scaling-for-complex-tasks#ran","syntology_url":"https://syntology.ai/paper/2504.00294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.00294"}},"official":null}},{"url":"/paper/video-r1-reinforcing-video-reasoning-in-mllms","slug":"video-r1-reinforcing-video-reasoning-in-mllms","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","date":"2025-03-27","arxiv_id":"2503.21776","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-r1-reinforcing-video-reasoning-in-mllms#ran","syntology_url":"https://syntology.ai/paper/2503.21776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21776"}},"official":{"repos":["tulerfeng/video-r1"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/mind-the-gap-benchmarking-spatial-reasoning","slug":"mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19707","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-gap-benchmarking-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19707"}},"official":{"repos":["stogiannidis/srbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/metaspatial-reinforcing-3d-spatial-reasoning","slug":"metaspatial-reinforcing-3d-spatial-reasoning","title":"MetaSpatial: Reinforcing 3D Spatial Reasoning in VLMs for the Metaverse","date":"2025-03-24","arxiv_id":"2503.18470","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/metaspatial-reinforcing-3d-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.18470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18470"}},"official":{"repos":["pzyseere/metaspatial"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sonata-self-supervised-learning-of-reliable","slug":"sonata-self-supervised-learning-of-reliable","title":"Sonata: Self-Supervised Learning of Reliable Point Representations","date":"2025-03-20","arxiv_id":"2503.16429","repositories_listed":2,"syntology":{"n":19,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/sonata-self-supervised-learning-of-reliable#ran","syntology_url":"https://syntology.ai/paper/2503.16429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16429"}},"official":{"repos":["Pointcept/Pointcept","facebookresearch/sonata"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/pointvla-injecting-the-3d-world-into-vision","slug":"pointvla-injecting-the-3d-world-into-vision","title":"PointVLA: Injecting the 3D World into Vision-Language-Action Models","date":"2025-03-10","arxiv_id":"2503.07511","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pointvla-injecting-the-3d-world-into-vision#ran","syntology_url":"https://syntology.ai/paper/2503.07511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07511"}},"official":null}},{"url":"/paper/why-is-spatial-reasoning-hard-for-vlms-an","slug":"why-is-spatial-reasoning-hard-for-vlms-an","title":"Why Is Spatial Reasoning Hard for VLMs? An Attention Mechanism Perspective on Focus Areas","date":"2025-03-03","arxiv_id":"2503.01773","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/why-is-spatial-reasoning-hard-for-vlms-an#ran","syntology_url":"https://syntology.ai/paper/2503.01773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01773"}},"official":{"repos":["shiqichen17/adaptvis"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/from-text-to-space-mapping-abstract-spatial","slug":"from-text-to-space-mapping-abstract-spatial","title":"From Text to Space: Mapping Abstract Spatial Models in LLMs during a Grid-World Navigation Task","date":"2025-02-23","arxiv_id":"2502.16690","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-text-to-space-mapping-abstract-spatial#ran","syntology_url":"https://syntology.ai/paper/2502.16690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16690"}},"official":{"repos":["mneuronico/griw-world-spatial-orientation-task"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied","slug":"cityeqa-a-hierarchical-llm-agent-on-embodied","title":"CityEQA: A Hierarchical LLM Agent on Embodied Question Answering Benchmark in City Space","date":"2025-02-18","arxiv_id":"2502.12532","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/cityeqa-a-hierarchical-llm-agent-on-embodied#ran","syntology_url":"https://syntology.ai/paper/2502.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.12532"}},"official":{"repos":["biluyong/cityeqa","tsinghua-fib-lab/CityEQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ivispar-an-interactive-visual-spatial","slug":"ivispar-an-interactive-visual-spatial","title":"iVISPAR -- An Interactive Visual-Spatial Reasoning Benchmark for VLMs","date":"2025-02-05","arxiv_id":"2502.03214","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ivispar-an-interactive-visual-spatial#ran","syntology_url":"https://syntology.ai/paper/2502.03214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03214"}},"official":{"repos":["SharkyBamboozle/iVISPAR"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-reasoning-to-adapt-large-language","slug":"enhancing-reasoning-to-adapt-large-language","title":"Enhancing Reasoning to Adapt Large Language Models for Domain-Specific Applications","date":"2025-02-05","arxiv_id":"2502.04384","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-reasoning-to-adapt-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.04384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04384"}},"official":{"repos":["wenboown/generative-ai-for-semiconductor-physical-design"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mapeval-a-map-based-evaluation-of-geo-spatial","slug":"mapeval-a-map-based-evaluation-of-geo-spatial","title":"MapEval: A Map-Based Evaluation of Geo-Spatial Reasoning in Foundation Models","date":"2024-12-31","arxiv_id":"2501.00316","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mapeval-a-map-based-evaluation-of-geo-spatial#ran","syntology_url":"https://syntology.ai/paper/2501.00316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00316"}},"official":{"repos":["MapEval/MapEval-API","MapEval/MapEval-Textual","MapEval/MapEval-Visual"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thinking-in-space-how-multimodal-large","slug":"thinking-in-space-how-multimodal-large","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","date":"2024-12-18","arxiv_id":"2412.14171","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thinking-in-space-how-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2412.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14171"}},"official":{"repos":["vision-x-nyu/thinking-in-space"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphere-a-hierarchical-evaluation-on-spatial","slug":"sphere-a-hierarchical-evaluation-on-spatial","title":"SPHERE: A Hierarchical Evaluation on Spatial Perception and Reasoning for Vision-Language Models","date":"2024-12-17","arxiv_id":"2412.12693","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphere-a-hierarchical-evaluation-on-spatial#ran","syntology_url":"https://syntology.ai/paper/2412.12693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12693"}},"official":{"repos":["zwenyu/SPHERE-VLM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citywalker-learning-embodied-urban-navigation","slug":"citywalker-learning-embodied-urban-navigation","title":"CityWalker: Learning Embodied Urban Navigation from Web-Scale Videos","date":"2024-11-26","arxiv_id":"2411.17820","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citywalker-learning-embodied-urban-navigation#ran","syntology_url":"https://syntology.ai/paper/2411.17820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17820"}},"official":{"repos":["ai4ce/CityWalker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/probing-the-limitations-of-multimodal","slug":"probing-the-limitations-of-multimodal","title":"Probing the limitations of multimodal language models for chemistry and materials research","date":"2024-11-25","arxiv_id":"2411.16955","repositories_listed":3,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/probing-the-limitations-of-multimodal#ran","syntology_url":"https://syntology.ai/paper/2411.16955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16955"}},"official":{"repos":["lamalab-org/mac-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/drivemllm-a-benchmark-for-spatial","slug":"drivemllm-a-benchmark-for-spatial","title":"DriveMLLM: A Benchmark for Spatial Understanding with Multimodal Large Language Models in Autonomous Driving","date":"2024-11-20","arxiv_id":"2411.13112","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drivemllm-a-benchmark-for-spatial#ran","syntology_url":"https://syntology.ai/paper/2411.13112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13112"}},"official":{"repos":["xiandaguo/drive-mllm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-navigation-with-vision-language","slug":"end-to-end-navigation-with-vision-language","title":"End-to-End Navigation with Vision Language Models: Transforming Spatial Reasoning into Question-Answering","date":"2024-11-08","arxiv_id":"2411.05755","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-navigation-with-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.05755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05755"}},"official":{"repos":["Jirl-upenn/VLMnav"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rocket-1-master-open-world-interaction-with","slug":"rocket-1-master-open-world-interaction-with","title":"ROCKET-1: Mastering Open-World Interaction with Visual-Temporal Context Prompting","date":"2024-10-23","arxiv_id":"2410.17856","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rocket-1-master-open-world-interaction-with#ran","syntology_url":"https://syntology.ai/paper/2410.17856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17856"}},"official":{"repos":["CraftJarvis/ROCKET-1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-vision-language-models-represent-space-and","slug":"do-vision-language-models-represent-space-and","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","date":"2024-10-22","arxiv_id":"2410.17385","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/do-vision-language-models-represent-space-and#ran","syntology_url":"https://syntology.ai/paper/2410.17385","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17385"}},"official":{"repos":["sled-group/COMFORT"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/locality-alignment-improves-vision-language","slug":"locality-alignment-improves-vision-language","title":"Locality Alignment Improves Vision-Language Models","date":"2024-10-14","arxiv_id":"2410.11087","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/locality-alignment-improves-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.11087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11087"}},"official":null}},{"url":"/paper/polymath-a-challenging-multi-modal","slug":"polymath-a-challenging-multi-modal","title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","date":"2024-10-06","arxiv_id":"2410.14702","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polymath-a-challenging-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2410.14702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14702"}},"official":{"repos":["polymathbenchmark/PolyMATH"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openkd-opening-prompt-diversity-for-zero-and","slug":"openkd-opening-prompt-diversity-for-zero-and","title":"OpenKD: Opening Prompt Diversity for Zero- and Few-shot Keypoint Detection","date":"2024-09-30","arxiv_id":"2409.19899","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/openkd-opening-prompt-diversity-for-zero-and#ran","syntology_url":"https://syntology.ai/paper/2409.19899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19899"}},"official":{"repos":["alanlusun/openkd"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-action-and-reasoning-centric-image","slug":"learning-action-and-reasoning-centric-image","title":"Learning Action and Reasoning-Centric Image Editing from Videos and Simulations","date":"2024-07-03","arxiv_id":"2407.03471","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-action-and-reasoning-centric-image#ran","syntology_url":"https://syntology.ai/paper/2407.03471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03471"}},"official":{"repos":["McGill-NLP/AURORA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/is-a-picture-worth-a-thousand-words-delving","slug":"is-a-picture-worth-a-thousand-words-delving","title":"Is A Picture Worth A Thousand Words? Delving Into Spatial Reasoning for Vision Language Models","date":"2024-06-21","arxiv_id":"2406.14852","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":5,"n_instrument":7,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":7,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/is-a-picture-worth-a-thousand-words-delving#ran","syntology_url":"https://syntology.ai/paper/2406.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14852"}},"official":{"repos":["jiayuww/SpatialEval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/citygpt-empowering-urban-spatial-cognition-of","slug":"citygpt-empowering-urban-spatial-cognition-of","title":"CityGPT: Empowering Urban Spatial Cognition of Large Language Models","date":"2024-06-20","arxiv_id":"2406.13948","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citygpt-empowering-urban-spatial-cognition-of#ran","syntology_url":"https://syntology.ai/paper/2406.13948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13948"}},"official":{"repos":["tsinghua-fib-lab/citygpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spatialbot-precise-spatial-understanding-with","slug":"spatialbot-precise-spatial-understanding-with","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","date":"2024-06-19","arxiv_id":"2406.13642","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spatialbot-precise-spatial-understanding-with#ran","syntology_url":"https://syntology.ai/paper/2406.13642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13642"}},"official":{"repos":["baai-dcai/spatialbot"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neuro-symbolic-training-for-reasoning-over","slug":"neuro-symbolic-training-for-reasoning-over","title":"Neuro-symbolic Training for Reasoning over Spatial Language","date":"2024-06-19","arxiv_id":"2406.13828","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/neuro-symbolic-training-for-reasoning-over#ran","syntology_url":"https://syntology.ai/paper/2406.13828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13828"}},"official":{"repos":["premsrit/SPARTUNQChain"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-of-reasoning-efficient-training-of-llm","slug":"flow-of-reasoning-efficient-training-of-llm","title":"Flow of Reasoning:Training LLMs for Divergent Problem Solving with Minimal Examples","date":"2024-06-09","arxiv_id":"2406.05673","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-of-reasoning-efficient-training-of-llm#ran","syntology_url":"https://syntology.ai/paper/2406.05673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05673"}},"official":{"repos":["yu-fangxu/for"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/getting-it-right-improving-spatial","slug":"getting-it-right-improving-spatial","title":"Getting it Right: Improving Spatial Consistency in Text-to-Image Models","date":"2024-04-01","arxiv_id":"2404.01197","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/getting-it-right-improving-spatial#ran","syntology_url":"https://syntology.ai/paper/2404.01197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01197"}},"official":{"repos":["SPRIGHT-T2I/SPRIGHT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/what-s-up-with-vision-language-models","slug":"what-s-up-with-vision-language-models","title":"What's \"up\" with vision-language models? Investigating their struggle with spatial reasoning","date":"2023-10-30","arxiv_id":"2310.19785","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-s-up-with-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2310.19785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19785"}},"official":{"repos":["amitakamath/whatsup_vlms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/vision-language-models-are-zero-shot-reward","slug":"vision-language-models-are-zero-shot-reward","title":"Vision-Language Models are Zero-Shot Reward Models for Reinforcement Learning","date":"2023-10-19","arxiv_id":"2310.12921","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vision-language-models-are-zero-shot-reward#ran","syntology_url":"https://syntology.ai/paper/2310.12921","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12921"}},"official":{"repos":["alignmentresearch/vlmrm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-large-language-models-be-good-path","slug":"can-large-language-models-be-good-path","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","date":"2023-10-05","arxiv_id":"2310.03249","repositories_listed":1,"syntology":{"n":28,"n_ran":28,"n_constructed":0,"n_ran_checked":28,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":28,"n_pointer_only":28,"phrase":"28 ran (of which 0 constructed an object rather than computing a result; 28 with no instrument failure: 0 honoured, 0 violated, 28 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-large-language-models-be-good-path#ran","syntology_url":"https://syntology.ai/paper/2310.03249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03249"}},"official":{"repos":["mohamedaghzal/llms-as-path-planners"],"state":"official (archive's flag): 28 ran","n_ran":28,"n_constructed":0,"n_ran_no_instrument_failure":28,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-baselines-with-visual-instruction","slug":"improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","arxiv_id":"2310.03744","repositories_listed":9,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/improved-baselines-with-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2310.03744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03744"}},"official":null}},{"url":"/paper/talk2bev-language-enhanced-bird-s-eye-view","slug":"talk2bev-language-enhanced-bird-s-eye-view","title":"Talk2BEV: Language-enhanced Bird's-eye View Maps for Autonomous Driving","date":"2023-10-03","arxiv_id":"2310.02251","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/talk2bev-language-enhanced-bird-s-eye-view#ran","syntology_url":"https://syntology.ai/paper/2310.02251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02251"}},"official":null}},{"url":"/paper/smartplay-a-benchmark-for-llms-as-intelligent","slug":"smartplay-a-benchmark-for-llms-as-intelligent","title":"SmartPlay: A Benchmark for LLMs as Intelligent Agents","date":"2023-10-02","arxiv_id":"2310.01557","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/smartplay-a-benchmark-for-llms-as-intelligent#ran","syntology_url":"https://syntology.ai/paper/2310.01557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01557"}},"official":{"repos":["microsoft/smartplay"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/droppos-pre-training-vision-transformers-by-1","slug":"droppos-pre-training-vision-transformers-by-1","title":"DropPos: Pre-Training Vision Transformers by Reconstructing Dropped Positions","date":"2023-09-07","arxiv_id":"2309.03576","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/droppos-pre-training-vision-transformers-by-1#ran","syntology_url":"https://syntology.ai/paper/2309.03576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03576"}},"official":{"repos":["haochen-wang409/droppos"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-universal-semantic-geometric-representation","slug":"a-universal-semantic-geometric-representation","title":"A Universal Semantic-Geometric Representation for Robotic Manipulation","date":"2023-06-18","arxiv_id":"2306.10474","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-universal-semantic-geometric-representation#ran","syntology_url":"https://syntology.ai/paper/2306.10474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10474"}},"official":{"repos":["TongZhangTHU/sgr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-instruction-tuning-1","slug":"visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":35,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 8 where Syntology's instrument failed) · 35 unverified","sample_list":"/paper/visual-instruction-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2304.08485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08485"}},"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["community","listed","named_in_paper","official"]}}},{"url":"/paper/gpt-4-technical-report-1","slug":"gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-technical-report-1#ran","syntology_url":"https://syntology.ai/paper/2303.08774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08774"}},"official":{"repos":["openai/evals"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/are-deep-neural-networks-smarter-than-second","slug":"are-deep-neural-networks-smarter-than-second","title":"Are Deep Neural Networks SMARTer than Second Graders?","date":"2022-12-20","arxiv_id":"2212.09993","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-deep-neural-networks-smarter-than-second#ran","syntology_url":"https://syntology.ai/paper/2212.09993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09993"}},"official":{"repos":["merlresearch/SMART"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/location-aware-self-supervised-transformers","slug":"location-aware-self-supervised-transformers","title":"Location-Aware Self-Supervised Transformers for Semantic Segmentation","date":"2022-12-05","arxiv_id":"2212.02400","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/location-aware-self-supervised-transformers#ran","syntology_url":"https://syntology.ai/paper/2212.02400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02400"}},"official":{"repos":["google-research/scenic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-spatial-reasoning","slug":"visual-spatial-reasoning","title":"Visual Spatial Reasoning","date":"2022-04-30","arxiv_id":"2205.00363","repositories_listed":4,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2205.00363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00363"}},"official":{"repos":["cambridgeltl/visual-spatial-reasoning","sohojoe/clip_visual-spatial-reasoning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/stepgame-a-new-benchmark-for-robust-multi-hop-1","slug":"stepgame-a-new-benchmark-for-robust-multi-hop-1","title":"StepGame: A New Benchmark for Robust Multi-Hop Spatial Reasoning in Texts","date":"2022-04-18","arxiv_id":"2204.08292","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stepgame-a-new-benchmark-for-robust-multi-hop-1#ran","syntology_url":"https://syntology.ai/paper/2204.08292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08292"}},"official":{"repos":["ZhengxiangShi/StepGame"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reclip-a-strong-zero-shot-baseline-for-1","slug":"reclip-a-strong-zero-shot-baseline-for-1","title":"ReCLIP: A Strong Zero-Shot Baseline for Referring Expression Comprehension","date":"2022-04-12","arxiv_id":"2204.05991","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reclip-a-strong-zero-shot-baseline-for-1#ran","syntology_url":"https://syntology.ai/paper/2204.05991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05991"}},"official":{"repos":["allenai/reclip"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/capturing-shape-information-with-multi-scale","slug":"capturing-shape-information-with-multi-scale","title":"Capturing Shape Information with Multi-Scale Topological Loss Terms for 3D Reconstruction","date":"2022-03-03","arxiv_id":"2203.01703","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/capturing-shape-information-with-multi-scale#ran","syntology_url":"https://syntology.ai/paper/2203.01703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01703"}},"official":{"repos":["marrlab/shapr_torch"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sornet-spatial-object-centric-representations","slug":"sornet-spatial-object-centric-representations","title":"SORNet: Spatial Object-Centric Representations for Sequential Manipulation","date":"2021-09-08","arxiv_id":"2109.03891","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sornet-spatial-object-centric-representations#ran","syntology_url":"https://syntology.ai/paper/2109.03891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03891"}},"official":{"repos":["wentaoyuan/sornet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/teaching-agents-how-to-map-spatial-reasoning","slug":"teaching-agents-how-to-map-spatial-reasoning","title":"Teaching Agents how to Map: Spatial Reasoning for Multi-Object Navigation","date":"2021-07-13","arxiv_id":"2107.06011","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/teaching-agents-how-to-map-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2107.06011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06011"}},"official":{"repos":["PierreMarza/teaching_agents_how_to_map"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/spartqa-a-textual-question-answering","slug":"spartqa-a-textual-question-answering","title":"SpartQA: : A Textual Question Answering Benchmark for Spatial Reasoning","date":"2021-04-12","arxiv_id":"2104.05832","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/spartqa-a-textual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2104.05832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.05832"}},"official":{"repos":["HLR/SpartQA_generation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/long-range-arena-a-benchmark-for-efficient-1","slug":"long-range-arena-a-benchmark-for-efficient-1","title":"Long Range Arena: A Benchmark for Efficient Transformers","date":"2020-11-08","arxiv_id":"2011.04006","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/long-range-arena-a-benchmark-for-efficient-1#ran","syntology_url":"https://syntology.ai/paper/2011.04006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.04006"}},"official":{"repos":["google-research/long-range-arena"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-and-reasoning-with-the-graph","slug":"learning-and-reasoning-with-the-graph","title":"Learning and Reasoning with the Graph Structure Representation in Robotic Surgery","date":"2020-07-07","arxiv_id":"2007.03357","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-and-reasoning-with-the-graph#ran","syntology_url":"https://syntology.ai/paper/2007.03357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03357"}},"official":{"repos":["mobarakol/Surgical_SceneGraph_Generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"url":"/paper/se-kge-a-location-aware-knowledge-graph","slug":"se-kge-a-location-aware-knowledge-graph","title":"SE-KGE: A Location-Aware Knowledge Graph Embedding Model for Geographic Question Answering and Spatial Semantic Lifting","date":"2020-04-25","arxiv_id":"2004.14171","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/se-kge-a-location-aware-knowledge-graph#ran","syntology_url":"https://syntology.ai/paper/2004.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14171"}},"official":null}},{"url":"/paper/vsgnet-spatial-attention-network-for","slug":"vsgnet-spatial-attention-network-for","title":"VSGNet: Spatial Attention Network for Detecting Human Object Interactions Using Graph Convolutions","date":"2020-03-11","arxiv_id":"2003.05541","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vsgnet-spatial-attention-network-for#ran","syntology_url":"https://syntology.ai/paper/2003.05541","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05541"}},"official":{"repos":["ASMIftekhar/VSGNet"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/floornet-a-unified-framework-for-floorplan","slug":"floornet-a-unified-framework-for-floorplan","title":"FloorNet: A Unified Framework for Floorplan Reconstruction from 3D Scans","date":"2018-03-31","arxiv_id":"1804.00090","repositories_listed":2,"syntology":{"n":17,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/floornet-a-unified-framework-for-floorplan#ran","syntology_url":"https://syntology.ai/paper/1804.00090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00090"}},"official":{"repos":["art-programmer/FloorNet"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/guesswhat-visual-object-discovery-through","slug":"guesswhat-visual-object-discovery-through","title":"GuessWhat?! Visual object discovery through multi-modal dialogue","date":"2016-11-23","arxiv_id":"1611.08481","repositories_listed":4,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/guesswhat-visual-object-discovery-through#ran","syntology_url":"https://syntology.ai/paper/1611.08481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.08481"}},"official":null}}],"record_sha256":"057605e6425eb3eba6d0ca032d91525991b5e546b36d2fc93b8cd45454487fb6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}