{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/task-planning/papers/ran/1","list_of":"/task/task-planning","task":"Task Planning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,35],"of":35,"counts":{"archive_papers_tagged":344,"with_a_code_link":100,"where_syntology_ran_a_sample":35,"not_listed_spam_title":0,"listed":344,"listed_where_code_ran":35,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":30,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":30,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/task-planning/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/gta1-gui-test-time-scaling-agent","slug":"gta1-gui-test-time-scaling-agent","title":"GTA1: GUI Test-time Scaling Agent","date":"2025-07-08","arxiv_id":"2507.05791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gta1-gui-test-time-scaling-agent#ran","syntology_url":"https://syntology.ai/paper/2507.05791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05791"}},"official":{"repos":["yan98/gta1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robopara-dual-arm-robot-planning-with","slug":"robopara-dual-arm-robot-planning-with","title":"RoboPARA: Dual-Arm Robot Planning with Parallel Allocation and Recomposition Across Tasks","date":"2025-06-07","arxiv_id":"2506.06683","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robopara-dual-arm-robot-planning-with#ran","syntology_url":"https://syntology.ai/paper/2506.06683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.06683"}},"official":null}},{"url":"/paper/llm-empowered-embodied-agent-for-memory","slug":"llm-empowered-embodied-agent-for-memory","title":"LLM-Empowered Embodied Agent for Memory-Augmented Task Planning in Household Robotics","date":"2025-04-30","arxiv_id":"2504.21716","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-empowered-embodied-agent-for-memory#ran","syntology_url":"https://syntology.ai/paper/2504.21716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21716"}},"official":{"repos":["marc1198/chat-hsr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-s2-a-compositional-generalist","slug":"agent-s2-a-compositional-generalist","title":"Agent S2: A Compositional Generalist-Specialist Framework for Computer Use Agents","date":"2025-04-01","arxiv_id":"2504.00906","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agent-s2-a-compositional-generalist#ran","syntology_url":"https://syntology.ai/paper/2504.00906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.00906"}},"official":{"repos":["simular-ai/agent-s"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robotouille-an-asynchronous-planning","slug":"robotouille-an-asynchronous-planning","title":"Robotouille: An Asynchronous Planning Benchmark for LLM Agents","date":"2025-02-06","arxiv_id":"2502.05227","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robotouille-an-asynchronous-planning#ran","syntology_url":"https://syntology.ai/paper/2502.05227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05227"}},"official":{"repos":["portal-cornell/robotouille"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vinci-a-real-time-embodied-smart-assistant","slug":"vinci-a-real-time-embodied-smart-assistant","title":"Vinci: A Real-time Embodied Smart Assistant based on Egocentric Vision-Language Model","date":"2024-12-30","arxiv_id":"2412.21080","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vinci-a-real-time-embodied-smart-assistant#ran","syntology_url":"https://syntology.ai/paper/2412.21080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.21080"}},"official":{"repos":["opengvlab/vinci"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/multi-modal-grounded-planning-and-efficient","slug":"multi-modal-grounded-planning-and-efficient","title":"Multi-Modal Grounded Planning and Efficient Replanning For Learning Embodied Agents with A Few Examples","date":"2024-12-23","arxiv_id":"2412.17288","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-modal-grounded-planning-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2412.17288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.17288"}},"official":{"repos":["snumprlab/flare"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safeagentbench-a-benchmark-for-safe-task","slug":"safeagentbench-a-benchmark-for-safe-task","title":"SafeAgentBench: A Benchmark for Safe Task Planning of Embodied LLM Agents","date":"2024-12-17","arxiv_id":"2412.13178","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safeagentbench-a-benchmark-for-safe-task#ran","syntology_url":"https://syntology.ai/paper/2412.13178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13178"}},"official":{"repos":["shengyin1224/safeagentbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generalizable-vision-language-robotic","slug":"towards-generalizable-vision-language-robotic","title":"Towards Generalizable Vision-Language Robotic Manipulation: A Benchmark and LLM-guided 3D Policy","date":"2024-10-02","arxiv_id":"2410.01345","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-generalizable-vision-language-robotic#ran","syntology_url":"https://syntology.ai/paper/2410.01345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01345"}},"official":{"repos":["vlc-robot/robot-3dlotus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/closed-loop-long-horizon-robotic-planning-via","slug":"closed-loop-long-horizon-robotic-planning-via","title":"Closed-Loop Long-Horizon Robotic Planning via Equilibrium Sequence Modeling","date":"2024-10-02","arxiv_id":"2410.01440","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/closed-loop-long-horizon-robotic-planning-via#ran","syntology_url":"https://syntology.ai/paper/2410.01440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01440"}},"official":{"repos":["singularity0104/equilibrium-planner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/riskawarebench-towards-evaluating-physical","slug":"riskawarebench-towards-evaluating-physical","title":"EARBench: Towards Evaluating Physical Risk Awareness for Task Planning of Foundation Model-based Embodied AI Agents","date":"2024-08-08","arxiv_id":"2408.04449","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/riskawarebench-towards-evaluating-physical#ran","syntology_url":"https://syntology.ai/paper/2408.04449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04449"}},"official":{"repos":["zihao-ai/eairiskbench","zihao-ai/earbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/disco-embodied-navigation-and-interaction-via","slug":"disco-embodied-navigation-and-interaction-via","title":"DISCO: Embodied Navigation and Interaction via Differentiable Scene Semantics and Dual-level Control","date":"2024-07-20","arxiv_id":"2407.14758","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/disco-embodied-navigation-and-interaction-via#ran","syntology_url":"https://syntology.ai/paper/2407.14758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14758"}},"official":{"repos":["allenxuuu/disco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentpoison-red-teaming-llm-agents-via","slug":"agentpoison-red-teaming-llm-agents-via","title":"AgentPoison: Red-teaming LLM Agents via Poisoning Memory or Knowledge Bases","date":"2024-07-17","arxiv_id":"2407.12784","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agentpoison-red-teaming-llm-agents-via#ran","syntology_url":"https://syntology.ai/paper/2407.12784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12784"}},"official":{"repos":["BillChan226/AgentPoison"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/details-make-a-difference-object-state","slug":"details-make-a-difference-object-state","title":"Details Make a Difference: Object State-Sensitive Neurorobotic Task Planning","date":"2024-06-14","arxiv_id":"2406.09988","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/details-make-a-difference-object-state#ran","syntology_url":"https://syntology.ai/paper/2406.09988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09988"}},"official":{"repos":["xiao-wen-sun/ossa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nyu-ctf-dataset-a-scalable-open-source","slug":"nyu-ctf-dataset-a-scalable-open-source","title":"NYU CTF Bench: A Scalable Open-Source Benchmark Dataset for Evaluating LLMs in Offensive Security","date":"2024-06-08","arxiv_id":"2406.05590","repositories_listed":5,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nyu-ctf-dataset-a-scalable-open-source#ran","syntology_url":"https://syntology.ai/paper/2406.05590","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05590"}},"official":{"repos":["nyu-llm-ctf/llm_ctf_automation","nyu-llm-ctf/llm_ctf_database","nyu-llm-ctf/nyu_ctf_bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tool-planner-dynamic-solution-tree-planning","slug":"tool-planner-dynamic-solution-tree-planning","title":"Tool-Planner: Task Planning with Clusters across Multiple Tools","date":"2024-06-06","arxiv_id":"2406.03807","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":1,"n_ran_checked":1,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tool-planner-dynamic-solution-tree-planning#ran","syntology_url":"https://syntology.ai/paper/2406.03807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03807"}},"official":{"repos":["OceannTwT/Tool-Planner"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-graph-learning-improve-task-planning","slug":"can-graph-learning-improve-task-planning","title":"Can Graph Learning Improve Planning in LLM-based Agents?","date":"2024-05-29","arxiv_id":"2405.19119","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-graph-learning-improve-task-planning#ran","syntology_url":"https://syntology.ai/paper/2405.19119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19119"}},"official":{"repos":["wxxshirley/gnn4taskplan"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-the-general-agent-capabilities-of","slug":"enhancing-the-general-agent-capabilities-of","title":"Enhancing the General Agent Capabilities of Low-Parameter LLMs through Tuning and Multi-Branch Reasoning","date":"2024-03-29","arxiv_id":"2403.19962","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/enhancing-the-general-agent-capabilities-of#ran","syntology_url":"https://syntology.ai/paper/2403.19962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19962"}},"official":{"repos":["haiv-lab/llm-tmbr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-3-large-language-model-based-task-and","slug":"llm-3-large-language-model-based-task-and","title":"LLM3:Large Language Model-based Task and Motion Planning with Motion Failure Reasoning","date":"2024-03-18","arxiv_id":"2403.11552","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-3-large-language-model-based-task-and#ran","syntology_url":"https://syntology.ai/paper/2403.11552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11552"}},"official":{"repos":["assassinws/llm-tamp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lota-bench-benchmarking-language-oriented","slug":"lota-bench-benchmarking-language-oriented","title":"LoTa-Bench: Benchmarking Language-oriented Task Planners for Embodied Agents","date":"2024-02-13","arxiv_id":"2402.08178","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lota-bench-benchmarking-language-oriented#ran","syntology_url":"https://syntology.ai/paper/2402.08178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08178"}},"official":{"repos":["lbaa2022/llmtaskplanning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/prismatic-vlms-investigating-the-design-space","slug":"prismatic-vlms-investigating-the-design-space","title":"Prismatic VLMs: Investigating the Design Space of Visually-Conditioned Language Models","date":"2024-02-12","arxiv_id":"2402.07865","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prismatic-vlms-investigating-the-design-space#ran","syntology_url":"https://syntology.ai/paper/2402.07865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07865"}},"official":{"repos":["tri-ml/prismatic-vlms","tri-ml/vlm-evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/trustagent-towards-safe-and-trustworthy-llm","slug":"trustagent-towards-safe-and-trustworthy-llm","title":"TrustAgent: Towards Safe and Trustworthy LLM-based Agents","date":"2024-02-02","arxiv_id":"2402.01586","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trustagent-towards-safe-and-trustworthy-llm#ran","syntology_url":"https://syntology.ai/paper/2402.01586","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01586"}},"official":{"repos":["agiresearch/trustagent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/small-llms-are-weak-tool-learners-a-multi-llm","slug":"small-llms-are-weak-tool-learners-a-multi-llm","title":"Small LLMs Are Weak Tool Learners: A Multi-LLM Agent","date":"2024-01-14","arxiv_id":"2401.07324","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/small-llms-are-weak-tool-learners-a-multi-llm#ran","syntology_url":"https://syntology.ai/paper/2401.07324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07324"}},"official":{"repos":["x-plug/multi-llm-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/egoplan-bench-benchmarking-egocentric","slug":"egoplan-bench-benchmarking-egocentric","title":"EgoPlan-Bench: Benchmarking Multimodal Large Language Models for Human-Level Planning","date":"2023-12-11","arxiv_id":"2312.06722","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/egoplan-bench-benchmarking-egocentric#ran","syntology_url":"https://syntology.ai/paper/2312.06722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06722"}},"official":{"repos":["chenyi99/egoplan"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-as-in-painting-a-diffusion-based","slug":"planning-as-in-painting-a-diffusion-based","title":"Planning as In-Painting: A Diffusion-Based Embodied Task Planning Framework for Environments under Uncertainty","date":"2023-12-02","arxiv_id":"2312.01097","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/planning-as-in-painting-a-diffusion-based#ran","syntology_url":"https://syntology.ai/paper/2312.01097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01097"}},"official":{"repos":["joeyy5588/planning-as-inpainting"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/kinematic-aware-prompting-for-generalizable","slug":"kinematic-aware-prompting-for-generalizable","title":"Kinematic-aware Prompting for Generalizable Articulated Object Manipulation with LLMs","date":"2023-11-06","arxiv_id":"2311.02847","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kinematic-aware-prompting-for-generalizable#ran","syntology_url":"https://syntology.ai/paper/2311.02847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02847"}},"official":{"repos":["gewu-lab/llm_articulated_object_manipulation","xwinks/llm_articulated_object_manipulation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/new-interaction-paradigm-for-complex-eda","slug":"new-interaction-paradigm-for-complex-eda","title":"New Interaction Paradigm for Complex EDA Software Leveraging GPT","date":"2023-07-27","arxiv_id":"2307.14740","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/new-interaction-paradigm-for-complex-eda#ran","syntology_url":"https://syntology.ai/paper/2307.14740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14740"}},"official":{"repos":["smarton-empower/smarton-ai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/getting-pwn-d-by-ai-penetration-testing-with","slug":"getting-pwn-d-by-ai-penetration-testing-with","title":"Getting pwn'd by AI: Penetration Testing with Large Language Models","date":"2023-07-24","arxiv_id":"2308.00121","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/getting-pwn-d-by-ai-penetration-testing-with#ran","syntology_url":"https://syntology.ai/paper/2308.00121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00121"}},"official":{"repos":["ipa-lab/hackingBuddyGPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embodied-task-planning-with-large-language","slug":"embodied-task-planning-with-large-language","title":"Embodied Task Planning with Large Language Models","date":"2023-07-04","arxiv_id":"2307.01848","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embodied-task-planning-with-large-language#ran","syntology_url":"https://syntology.ai/paper/2307.01848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.01848"}},"official":{"repos":["Gary3410/TaPA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/integrating-action-knowledge-and-llms-for","slug":"integrating-action-knowledge-and-llms-for","title":"Integrating Action Knowledge and LLMs for Task Planning and Situation Handling in Open Worlds","date":"2023-05-27","arxiv_id":"2305.17590","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/integrating-action-knowledge-and-llms-for#ran","syntology_url":"https://syntology.ai/paper/2305.17590","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17590"}},"official":{"repos":["yding25/GPT-Planner"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hugginggpt-solving-ai-tasks-with-chatgpt-and","slug":"hugginggpt-solving-ai-tasks-with-chatgpt-and","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","date":"2023-03-30","arxiv_id":"2303.17580","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hugginggpt-solving-ai-tasks-with-chatgpt-and#ran","syntology_url":"https://syntology.ai/paper/2303.17580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17580"}},"official":{"repos":["microsoft/JARVIS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-framework-to-generate-neurosymbolic-pddl","slug":"a-framework-to-generate-neurosymbolic-pddl","title":"A Framework for Neurosymbolic Robot Action Planning using Large Language Models","date":"2023-03-01","arxiv_id":"2303.00438","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-framework-to-generate-neurosymbolic-pddl#ran","syntology_url":"https://syntology.ai/paper/2303.00438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.00438"}},"official":{"repos":["alessiocpt/teriyaki"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimedia-generative-script-learning-for","slug":"multimedia-generative-script-learning-for","title":"Multimedia Generative Script Learning for Task Planning","date":"2022-08-25","arxiv_id":"2208.12306","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multimedia-generative-script-learning-for#ran","syntology_url":"https://syntology.ai/paper/2208.12306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.12306"}},"official":{"repos":["EagleW/Multimedia-Generative-Script-Learning-for-Task-Planning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sornet-spatial-object-centric-representations","slug":"sornet-spatial-object-centric-representations","title":"SORNet: Spatial Object-Centric Representations for Sequential Manipulation","date":"2021-09-08","arxiv_id":"2109.03891","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sornet-spatial-object-centric-representations#ran","syntology_url":"https://syntology.ai/paper/2109.03891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.03891"}},"official":{"repos":["wentaoyuan/sornet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"cf614acb8e49842b991b89770c8ae8ed78851ba5c2ff0b2647909f2538ec1cf7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}