{"url":"/task/action-generation","name":"Action Generation","slug":"action-generation","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":111,"papers_with_code":49,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/humanact12","name":"HumanAct12","full_name":"","num_papers_in_archive":37},{"url":"/dataset/phspd","name":"PHSPD","full_name":"Polarization Human Shape and Pose Dataset","num_papers_in_archive":2},{"url":"/dataset/llmafia","name":"LLMafia","full_name":"","num_papers_in_archive":1},{"url":"/dataset/two4two","name":"Two4Two","full_name":"A Synthetic Dataset For Controlled Experiments","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/human-action-generation","name":"Human action generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":49,"tagged_in_all":111,"items":[{"url":"/paper/mapping-instructions-to-actions-in-3d","title":"Mapping Instructions to Actions in 3D Environments with Visual Goal Prediction","date":"2018-09-04","arxiv_id":"1809.00786","repositories_listed":5,"syntology":null},{"url":"/paper/infigui-r1-advancing-multimodal-gui-agents","title":"InfiGUI-R1: Advancing Multimodal GUI Agents from Reactive Actors to Deliberative Reasoners","date":"2025-04-19","arxiv_id":"2504.14239","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/flow-q-learning","title":"Flow Q-Learning","date":"2025-02-04","arxiv_id":"2502.02538","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/autocrawler-a-progressive-understanding-web","title":"AutoScraper: A Progressive Understanding Web Agent for Web Scraper Generation","date":"2024-04-19","arxiv_id":"2404.12753","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/human-action-generation-with-generative","title":"Human Action Generation with Generative Adversarial Networks","date":"2018-05-26","arxiv_id":"1805.10416","repositories_listed":2,"syntology":null},{"url":"/paper/worldvla-towards-autoregressive-action-world","title":"WorldVLA: Towards Autoregressive Action World Model","date":"2025-06-26","arxiv_id":"2506.21539","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":4}},{"url":"/paper/parallels-between-vla-model-post-training-and","title":"Parallels Between VLA Model Post-Training and Human Motor Learning: Progress, Challenges, and Trends","date":"2025-06-26","arxiv_id":"2506.20966","repositories_listed":1,"syntology":null},{"url":"/paper/autovla-a-vision-language-action-model-for","title":"AutoVLA: A Vision-Language-Action Model for End-to-End Autonomous Driving with Adaptive Reasoning and Reinforcement Fine-Tuning","date":"2025-06-16","arxiv_id":"2506.13757","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/time-to-talk-llm-agents-for-asynchronous","title":"Time to Talk: LLM Agents for Asynchronous Group Communication in Mafia Games","date":"2025-06-05","arxiv_id":"2506.05309","repositories_listed":1,"syntology":null},{"url":"/paper/owmm-agent-open-world-mobile-manipulation","title":"OWMM-Agent: Open World Mobile Manipulation With Multi-modal Agentic Data Synthesis","date":"2025-06-04","arxiv_id":"2506.04217","repositories_listed":1,"syntology":null},{"url":"/paper/star-learning-diverse-robot-skill","title":"STAR: Learning Diverse Robot Skill Abstractions through Rotation-Augmented Vector Quantization","date":"2025-06-04","arxiv_id":"2506.03863","repositories_listed":1,"syntology":null},{"url":"/paper/smolvla-a-vision-language-action-model-for","title":"SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics","date":"2025-06-02","arxiv_id":"2506.01844","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/distilling-llm-agent-into-small-models-with","title":"Distilling LLM Agent into Small Models with Retrieval and Code Tools","date":"2025-05-23","arxiv_id":"2505.17612","repositories_listed":1,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/llm-explorer-towards-efficient-and-affordable","title":"LLM-Explorer: Towards Efficient and Affordable LLM-based Exploration for Mobile Apps","date":"2025-05-15","arxiv_id":"2505.10593","repositories_listed":1,"syntology":null},{"url":"/paper/train-a-multi-task-diffusion-policy-on","title":"Mini Diffuser: Fast Multi-task Diffusion Policy Training Using Two-level Mini-batches","date":"2025-05-14","arxiv_id":"2505.09430","repositories_listed":1,"syntology":null},{"url":"/paper/prior-does-matter-visual-navigation-via","title":"Prior Does Matter: Visual Navigation via Denoising Diffusion Bridge Models","date":"2025-04-14","arxiv_id":"2504.10041","repositories_listed":1,"syntology":null},{"url":"/paper/agent-models-internalizing-chain-of-action","title":"Agent models: Internalizing Chain-of-Action Generation into Reasoning models","date":"2025-03-09","arxiv_id":"2503.06580","repositories_listed":1,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/litewebagent-the-open-source-suite-for-vlm","title":"LiteWebAgent: The Open-Source Suite for VLM-Based Web-Agent Applications","date":"2025-03-04","arxiv_id":"2503.02950","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-a-good-diffusion-planner-for","title":"What Makes a Good Diffusion Planner for Decision Making?","date":"2025-03-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-vision-language-action-models","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","date":"2025-02-27","arxiv_id":"2502.19645","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/pmat-optimizing-action-generation-order-in","title":"PMAT: Optimizing Action Generation Order in Multi-Agent Reinforcement Learning","date":"2025-02-23","arxiv_id":"2502.16496","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-multi-robot-systems","title":"Large Language Models for Multi-Robot Systems: A Survey","date":"2025-02-06","arxiv_id":"2502.03814","repositories_listed":1,"syntology":null},{"url":"/paper/large-action-models-from-inception-to","title":"Large Action Models: From Inception to Implementation","date":"2024-12-13","arxiv_id":"2412.10047","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-vision-language-action-models-on","title":"Benchmarking Vision, Language, & Action Models on Robotic Learning Tasks","date":"2024-11-04","arxiv_id":"2411.05821","repositories_listed":1,"syntology":null},{"url":"/paper/seg2act-global-context-aware-action","title":"Seg2Act: Global Context-aware Action Generation for Document Logical Structuring","date":"2024-10-09","arxiv_id":"2410.06802","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":8}},{"url":"/paper/affordance-based-robot-manipulation-with-flow","title":"Affordance-based Robot Manipulation with Flow Matching","date":"2024-09-02","arxiv_id":"2409.01083","repositories_listed":1,"syntology":null},{"url":"/paper/epo-hierarchical-llm-agents-with-environment","title":"EPO: Hierarchical LLM Agents with Environment Preference Optimization","date":"2024-08-28","arxiv_id":"2408.16090","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/solving-robotics-problems-in-zero-shot-with","title":"Wonderful Team: Zero-Shot Physical Task Planning with Visual LLMs","date":"2024-07-26","arxiv_id":"2407.19094","repositories_listed":1,"syntology":null},{"url":"/paper/echoreel-enhancing-action-generation-of","title":"AICL: Action In-Context Learning for Video Diffusion Model","date":"2024-03-18","arxiv_id":"2403.11535","repositories_listed":1,"syntology":null},{"url":"/paper/pokellmon-a-human-parity-agent-for-pokemon","title":"PokeLLMon: A Human-Parity Agent for Pokemon Battles with Large Language Models","date":"2024-02-02","arxiv_id":"2402.01118","repositories_listed":1,"syntology":null}],"syntology_records":11,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}