{"url":"/task/robot-manipulation","name":"Robot Manipulation","slug":"robot-manipulation","description_markdown":null,"categories":[{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":430,"papers_with_code":154,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":12,"subtasks":3,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/robot-manipulation-on-calvin","slug":"robot-manipulation-on-calvin","dataset":"CALVIN","dataset_url":"/dataset/calvin-composing-actions-from-language-and","rows_in_archive":19,"metrics":["avg. sequence length (D to D)"],"first_row_in_archive_order":{"model":"DreamVLA","paper_title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","paper_url":"/paper/dreamvla-a-vision-language-action-model-1","paper_date":"2025-07-06","arxiv_id":"2507.04447","code_links":[{"title":"Zhangwenyao1/DreamVLA","url":"https://github.com/Zhangwenyao1/DreamVLA"}],"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}}},{"leaderboard":"/sota/robot-manipulation-on-rlbench","slug":"robot-manipulation-on-rlbench","dataset":"RLBench","dataset_url":"/dataset/rlbench","rows_in_archive":18,"metrics":["Succ. Rate (18 tasks, 100 demo/task)","Succ. Rate (18 tasks, 10 demo/task)","Training Time (V100 x 8 x day)","Training Time (A100 x hour)","Succ. Rate (10 tasks, 100 demos/task)","Succ. Rate (74 tasks, 100 demos/task)","Inference Speed (fps)","Input Image Size"],"first_row_in_archive_order":{"model":"EquAct","paper_title":"EquAct: An SE(3)-Equivariant Multi-Task Transformer for Open-Loop Robotic Manipulation","paper_url":null,"paper_date":"2025-05-27","arxiv_id":"2505.21351","code_links":[],"syntology":null}},{"leaderboard":"/sota/robot-manipulation-on-simpler-env","slug":"robot-manipulation-on-simpler-env","dataset":"SimplerEnv-Google Robot","dataset_url":"/dataset/simpler-env","rows_in_archive":9,"metrics":["Visual Matching","Visual Matching-Pick Coke Can","Visual Matching-Move Near","Visual Matching-Open/Close Drawer","Variant Aggregation","Variant Aggregation-Pick Coke Can","Variant Aggregation-Move Near","Variant Aggregation-Open/Close Drawer"],"first_row_in_archive_order":{"model":"SoFar","paper_title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","paper_url":"/paper/sofar-language-grounded-orientation-bridges","paper_date":"2025-02-18","arxiv_id":"2502.13143","code_links":[{"title":"qizekun/SoFar","url":"https://github.com/qizekun/SoFar"},{"title":"zhangwenyao1/open6dor_v2_execution","url":"https://github.com/zhangwenyao1/open6dor_v2_execution"}],"syntology":null}},{"leaderboard":"/sota/robot-manipulation-on-mimicgen","slug":"robot-manipulation-on-mimicgen","dataset":"MimicGen","dataset_url":null,"rows_in_archive":7,"metrics":["Succ. Rate (12 tasks, 100 demo/task)","Succ. Rate (12 tasks, 1000 demo/task)","Succ. Rate (12 tasks, 200 demo/task)"],"first_row_in_archive_order":{"model":"SDP","paper_title":"SE(3)-Equivariant Diffusion Policy in Spherical Fourier Space","paper_url":"/paper/se-3-equivariant-diffusion-policy-in-1","paper_date":"2025-07-02","arxiv_id":"2507.01723","code_links":[{"title":"amazon-science/Spherical_Diffusion_Policy","url":"https://github.com/amazon-science/Spherical_Diffusion_Policy"}],"syntology":{"n":14,"n_ran":7,"n_unverified":7,"n_pointer_only":0}}},{"leaderboard":"/sota/robot-manipulation-on-simplerenv-widow-x","slug":"robot-manipulation-on-simplerenv-widow-x","dataset":"SimplerEnv-Widow X","dataset_url":"/dataset/simplerenv-widow-x","rows_in_archive":7,"metrics":["Average","Put Spoon on Towel","Put Carrot on Plate","Stack Green Block on Yellow Block","Put Eggplant  in Yellow Basket","Put Eggplant in Yellow Basket"],"first_row_in_archive_order":{"model":"SoFar","paper_title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","paper_url":"/paper/sofar-language-grounded-orientation-bridges","paper_date":"2025-02-18","arxiv_id":"2502.13143","code_links":[{"title":"qizekun/SoFar","url":"https://github.com/qizekun/SoFar"},{"title":"zhangwenyao1/open6dor_v2_execution","url":"https://github.com/zhangwenyao1/open6dor_v2_execution"}],"syntology":null}}],"datasets":[{"url":"/dataset/rlbench","name":"RLBench","full_name":"","num_papers_in_archive":167},{"url":"/dataset/calvin-composing-actions-from-language-and","name":"CALVIN","full_name":"Composing Actions from Language and Vision","num_papers_in_archive":65},{"url":"/dataset/maniskill2","name":"ManiSkill2","full_name":"","num_papers_in_archive":40},{"url":"/dataset/simpler-env","name":"SimplerEnv-Google Robot","full_name":"Evaluating Real-World Robot Manipulation Policies in Simulation","num_papers_in_archive":10},{"url":"/dataset/simplerenv-widow-x","name":"SimplerEnv-Widow X","full_name":"SimplerEnv: Simulated Manipulation Policy Evaluation Environments for Real Robot Setups","num_papers_in_archive":8},{"url":"/dataset/robosuite-benchmark","name":"robosuite Benchmark","full_name":"robosuite Benchmark","num_papers_in_archive":7},{"url":"/dataset/ycb-slide","name":"YCB-Slide","full_name":"YCB-Slide: A tactile interaction dataset","num_papers_in_archive":7},{"url":"/dataset/furniturebench-diffik-sim-demos","name":"FurnitureBench DiffIK sim demos","full_name":"","num_papers_in_archive":1},{"url":"/dataset/manipulatesound","name":"ManipulateSound","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mikasa-robo-dataset","name":"MIKASA-Robo Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/motif-1k","name":"MotIF-1K","full_name":"","num_papers_in_archive":1},{"url":"/dataset/pick-screw","name":"pick_screw","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/contact-rich-manipulation","name":"Contact-rich Manipulation"},{"url":"/task/deformable-object-manipulation","name":"Deformable Object Manipulation"},{"url":"/task/robot-manipulation-generalization","name":"Robot Manipulation Generalization"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":154,"tagged_in_all":430,"items":[{"url":"/paper/openvla-an-open-source-vision-language-action","title":"OpenVLA: An Open-Source Vision-Language-Action Model","date":"2024-06-13","arxiv_id":"2406.09246","repositories_listed":3,"syntology":{"n":10,"n_ran":2,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/unleashing-large-scale-video-generative-pre","title":"Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation","date":"2023-12-20","arxiv_id":"2312.13139","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/magma-a-foundation-model-for-multimodal-ai","title":"Magma: A Foundation Model for Multimodal AI Agents","date":"2025-02-18","arxiv_id":"2502.13130","repositories_listed":2,"syntology":{"n":16,"n_ran":2,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/sofar-language-grounded-orientation-bridges","title":"SoFar: Language-Grounded Orientation Bridges Spatial Reasoning and Object Manipulation","date":"2025-02-18","arxiv_id":"2502.13143","repositories_listed":2,"syntology":null},{"url":"/paper/act3d-infinite-resolution-action-detection","title":"Act3D: 3D Feature Field Transformers for Multi-Task Robotic Manipulation","date":"2023-06-30","arxiv_id":"2306.17817","repositories_listed":2,"syntology":null},{"url":"/paper/libero-benchmarking-knowledge-transfer-for","title":"LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning","date":"2023-06-05","arxiv_id":"2306.03310","repositories_listed":2,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/vima-general-robot-manipulation-with","title":"VIMA: General Robot Manipulation with Multimodal Prompts","date":"2022-10-06","arxiv_id":"2210.03094","repositories_listed":2,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/instruction-driven-history-aware-policies-for","title":"Instruction-driven history-aware policies for robotic manipulations","date":"2022-09-11","arxiv_id":"2209.04899","repositories_listed":2,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/reward-uncertainty-for-exploration-in-1","title":"Reward Uncertainty for Exploration in Preference-based Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12401","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/what-matters-in-language-conditioned-robotic","title":"What Matters in Language Conditioned Robotic Imitation Learning over Unstructured Data","date":"2022-04-13","arxiv_id":"2204.06252","repositories_listed":2,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/task-focused-few-shot-object-detection-for","title":"Mobile Robot Manipulation using Pure Object Detection","date":"2022-01-28","arxiv_id":"2201.12437","repositories_listed":2,"syntology":null},{"url":"/paper/learning-3d-dynamic-scene-representations-for","title":"Learning 3D Dynamic Scene Representations for Robot Manipulation","date":"2020-11-03","arxiv_id":"2011.01968","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-robotic","title":"Reinforcement Learning for Robotic Manipulation using Simulated Locomotion Demonstrations","date":"2019-10-16","arxiv_id":"1910.07294","repositories_listed":2,"syntology":null},{"url":"/paper/silhonet-an-rgb-method-for-6d-object-pose","title":"SilhoNet: An RGB Method for 6D Object Pose Estimation","date":"2018-09-18","arxiv_id":"1809.06893","repositories_listed":2,"syntology":null},{"url":"/paper/deepim-deep-iterative-matching-for-6d-pose","title":"DeepIM: Deep Iterative Matching for 6D Pose Estimation","date":"2018-03-31","arxiv_id":"1804.00175","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/dreamvla-a-vision-language-action-model-1","title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","date":"2025-07-06","arxiv_id":"2507.04447","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/3dflowaction-learning-cross-embodiment","title":"3DFlowAction: Learning Cross-Embodiment Manipulation from 3D Flow World Model","date":"2025-06-06","arxiv_id":"2506.06199","repositories_listed":1,"syntology":null},{"url":"/paper/rlvr-world-training-world-models-with","title":"RLVR-World: Training World Models with Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.13934","repositories_listed":1,"syntology":{"n":13,"n_ran":1,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/train-a-multi-task-diffusion-policy-on","title":"Mini Diffuser: Fast Multi-task Diffusion Policy Training Using Two-level Mini-batches","date":"2025-05-14","arxiv_id":"2505.09430","repositories_listed":1,"syntology":null},{"url":"/paper/from-seeing-to-doing-bridging-reasoning-and","title":"From Seeing to Doing: Bridging Reasoning and Decision for Robotic Manipulation","date":"2025-05-13","arxiv_id":"2505.08548","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/univla-learning-to-act-anywhere-with-task","title":"UniVLA: Learning to Act Anywhere with Task-centric Latent Actions","date":"2025-05-09","arxiv_id":"2505.06111","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/openhelix-a-short-survey-empirical-analysis","title":"OpenHelix: A Short Survey, Empirical Analysis, and Open-Source Dual-System VLA Model for Robotic Manipulation","date":"2025-05-06","arxiv_id":"2505.03912","repositories_listed":1,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/sim2real-transfer-for-vision-based-grasp","title":"Sim2Real Transfer for Vision-Based Grasp Verification","date":"2025-05-05","arxiv_id":"2505.03046","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-fusion-and-vision-language-models","title":"Multimodal Fusion and Vision-Language Models: A Survey for Robot Vision","date":"2025-04-03","arxiv_id":"2504.02477","repositories_listed":1,"syntology":null},{"url":"/paper/autoeval-autonomous-evaluation-of-generalist","title":"AutoEval: Autonomous Evaluation of Generalist Robot Manipulation Policies in the Real World","date":"2025-03-31","arxiv_id":"2503.24278","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/dita-scaling-diffusion-transformer-for","title":"Dita: Scaling Diffusion Transformer for Generalist Vision-Language-Action Policy","date":"2025-03-25","arxiv_id":"2503.19757","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/phystwin-physics-informed-reconstruction-and","title":"PhysTwin: Physics-Informed Reconstruction and Simulation of Deformable Objects from Videos","date":"2025-03-23","arxiv_id":"2503.17973","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/combined-physics-and-event-camera-simulator","title":"Combined Physics and Event Camera Simulator for Slip Detection","date":"2025-03-05","arxiv_id":"2503.04838","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-fine-grained-hand-object-dynamics","title":"Modeling Fine-Grained Hand-Object Dynamics for Egocentric Video Representation Learning","date":"2025-03-02","arxiv_id":"2503.00986","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/chatvla-unified-multimodal-understanding-and","title":"ChatVLA: Unified Multimodal Understanding and Robot Control with Vision-Language-Action Model","date":"2025-02-20","arxiv_id":"2502.14420","repositories_listed":1,"syntology":null}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}