{"url":"/task/task-planning","name":"Task Planning","slug":"task-planning","description_markdown":null,"categories":[{"name":"Reasoning","url":"/area/reasoning"},{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":344,"papers_with_code":100,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/plancraft","name":"Plancraft","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/robot-task-planning","name":"Robot Task Planning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":100,"tagged_in_all":344,"items":[{"url":"/paper/nyu-ctf-dataset-a-scalable-open-source","title":"NYU CTF Bench: A Scalable Open-Source Benchmark Dataset for Evaluating LLMs in Offensive Security","date":"2024-06-08","arxiv_id":"2406.05590","repositories_listed":5,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/prismatic-vlms-investigating-the-design-space","title":"Prismatic VLMs: Investigating the Design Space of Visually-Conditioned Language Models","date":"2024-02-12","arxiv_id":"2402.07865","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":4}},{"url":"/paper/a-comparison-of-prompt-engineering-techniques","title":"A Comparison of Prompt Engineering Techniques for Task Planning and Execution in Service Robotics","date":"2024-10-30","arxiv_id":"2410.22997","repositories_listed":2,"syntology":null},{"url":"/paper/riskawarebench-towards-evaluating-physical","title":"EARBench: Towards Evaluating Physical Risk Awareness for Task Planning of Foundation Model-based Embodied AI Agents","date":"2024-08-08","arxiv_id":"2408.04449","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/personal-llm-agents-insights-and-survey-about","title":"Personal LLM Agents: Insights and Survey about the Capability, Efficiency and Security","date":"2024-01-10","arxiv_id":"2401.05459","repositories_listed":2,"syntology":null},{"url":"/paper/kinematic-aware-prompting-for-generalizable","title":"Kinematic-aware Prompting for Generalizable Articulated Object Manipulation with LLMs","date":"2023-11-06","arxiv_id":"2311.02847","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/isr-llm-iterative-self-refined-large-language","title":"ISR-LLM: Iterative Self-Refined Large Language Model for Long-Horizon Sequential Task Planning","date":"2023-08-26","arxiv_id":"2308.13724","repositories_listed":2,"syntology":null},{"url":"/paper/plansys2-a-planning-system-framework-for-ros2","title":"PlanSys2: A Planning System Framework for ROS2","date":"2021-07-01","arxiv_id":"2107.00376","repositories_listed":2,"syntology":null},{"url":"/paper/gta1-gui-test-time-scaling-agent","title":"GTA1: GUI Test-time Scaling Agent","date":"2025-07-08","arxiv_id":"2507.05791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/a-comprehensive-survey-of-deep-research","title":"A Comprehensive Survey of Deep Research: Systems, Methodologies, and Applications","date":"2025-06-14","arxiv_id":"2506.12594","repositories_listed":1,"syntology":null},{"url":"/paper/prime-the-search-using-large-language-models","title":"Prime the search: Using large language models for guiding geometric task and motion planning by warm-starting tree search","date":"2025-06-08","arxiv_id":"2506.07062","repositories_listed":1,"syntology":null},{"url":"/paper/flysearch-exploring-how-vision-language","title":"FlySearch: Exploring how vision-language models explore","date":"2025-06-03","arxiv_id":"2506.02896","repositories_listed":1,"syntology":null},{"url":"/paper/bedi-a-comprehensive-benchmark-for-evaluating","title":"BEDI: A Comprehensive Benchmark for Evaluating Embodied Agents on UAVs","date":"2025-05-23","arxiv_id":"2505.18229","repositories_listed":1,"syntology":null},{"url":"/paper/craken-cybersecurity-llm-agent-with-knowledge","title":"CRAKEN: Cybersecurity LLM Agent with Knowledge-Based Execution","date":"2025-05-21","arxiv_id":"2505.17107","repositories_listed":1,"syntology":null},{"url":"/paper/apex-empowering-llms-with-physics-based-task","title":"APEX: Empowering LLMs with Physics-Based Task Planning for Real-time Insight","date":"2025-05-20","arxiv_id":"2505.13921","repositories_listed":1,"syntology":null},{"url":"/paper/achieving-scalable-robot-autonomy-via","title":"Achieving Scalable Robot Autonomy via neurosymbolic planning using lightweight local LLM","date":"2025-05-13","arxiv_id":"2505.08492","repositories_listed":1,"syntology":null},{"url":"/paper/llm-empowered-embodied-agent-for-memory","title":"LLM-Empowered Embodied Agent for Memory-Augmented Task Planning in Household Robotics","date":"2025-04-30","arxiv_id":"2504.21716","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/enhancing-llm-based-agents-via-global","title":"Enhancing LLM-Based Agents via Global Planning and Hierarchical Execution","date":"2025-04-23","arxiv_id":"2504.16563","repositories_listed":1,"syntology":null},{"url":"/paper/agent-s2-a-compositional-generalist","title":"Agent S2: A Compositional Generalist-Specialist Framework for Computer Use Agents","date":"2025-04-01","arxiv_id":"2504.00906","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/data-agnostic-robotic-long-horizon","title":"Data-Agnostic Robotic Long-Horizon Manipulation with Vision-Language-Guided Closed-Loop Feedback","date":"2025-03-27","arxiv_id":"2503.21969","repositories_listed":1,"syntology":null},{"url":"/paper/llm-map-bimanual-robot-task-planning-using","title":"LLM+MAP: Bimanual Robot Task Planning using Large Language Models and Planning Domain Definition Language","date":"2025-03-21","arxiv_id":"2503.17309","repositories_listed":1,"syntology":null},{"url":"/paper/graphormer-guided-task-planning-beyond-static","title":"Graphormer-Guided Task Planning: Beyond Static Rules with LLM Safety Perception","date":"2025-03-10","arxiv_id":"2503.06866","repositories_listed":1,"syntology":null},{"url":"/paper/safe-llm-controlled-robots-with-formal","title":"Safe LLM-Controlled Robots with Formal Guarantees via Reachability Analysis","date":"2025-03-05","arxiv_id":"2503.03911","repositories_listed":1,"syntology":null},{"url":"/paper/clea-closed-loop-embodied-agent-for-enhancing","title":"CLEA: Closed-Loop Embodied Agent for Enhancing Task Execution in Dynamic Environments","date":"2025-03-02","arxiv_id":"2503.00729","repositories_listed":1,"syntology":null},{"url":"/paper/mrbtp-efficient-multi-robot-behavior-tree","title":"MRBTP: Efficient Multi-Robot Behavior Tree Planning and Collaboration","date":"2025-02-25","arxiv_id":"2502.18072","repositories_listed":1,"syntology":null},{"url":"/paper/plan-over-graph-towards-parallelable-llm","title":"Plan-over-Graph: Towards Parallelable LLM Agent Schedule","date":"2025-02-20","arxiv_id":"2502.14563","repositories_listed":1,"syntology":null},{"url":"/paper/navrag-generating-user-demand-instructions","title":"NavRAG: Generating User Demand Instructions for Embodied Navigation through Retrieval-Augmented LLM","date":"2025-02-16","arxiv_id":"2502.11142","repositories_listed":1,"syntology":null},{"url":"/paper/d-cipher-dynamic-collaborative-intelligent","title":"D-CIPHER: Dynamic Collaborative Intelligent Multi-Agent System with Planner and Heterogeneous Executors for Offensive Security","date":"2025-02-15","arxiv_id":"2502.10931","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-multi-robot-systems","title":"Large Language Models for Multi-Robot Systems: A Survey","date":"2025-02-06","arxiv_id":"2502.03814","repositories_listed":1,"syntology":null},{"url":"/paper/robotouille-an-asynchronous-planning","title":"Robotouille: An Asynchronous Planning Benchmark for LLM Agents","date":"2025-02-06","arxiv_id":"2502.05227","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}