{"url":"/dataset/calvin-composing-actions-from-language-and","name":"CALVIN","full_name":"Composing Actions from Language and Vision","description_markdown":"CALVIN (Composing Actions from Language and Vision), is an open-source simulated benchmark to learn long-horizon language-conditioned robot manipulation tasks.","description_withheld":null,"homepage":"https://github.com/mees/calvin","introduced_date":"2021-12-06","introduced_date_note":null,"introduced_by":{"paper":"/paper/calvin-a-benchmark-for-language-conditioned","title":"CALVIN: A Benchmark for Language-Conditioned Policy Learning for Long-Horizon Robot Manipulation Tasks","first_author":"Oier Mees","url":null},"license":{"name":"MIT License","url":null},"modalities":[],"tasks":[{"name":"Robot Manipulation","url":"/task/robot-manipulation","datasets_with_task":"/datasets/task/robot-manipulation"},{"name":"Success Rate (5 task-horizon)","url":"/task/success-rate-5-task-horizon","datasets_with_task":"/datasets/task/success-rate-5-task-horizon"},{"name":"Avg. sequence length","url":"/task/avg-sequence-length","datasets_with_task":"/datasets/task/avg-sequence-length"},{"name":"Zero-shot Generalization","url":"/task/zero-shot-generalization","datasets_with_task":"/datasets/task/zero-shot-generalization"}],"languages":[],"variants":["CALVIN"],"data_loaders":[{"repo":"https://github.com/mees/calvin","url":"https://github.com/mees/calvin","frameworks":[]}],"num_papers_in_archive":65,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/robot-manipulation-on-calvin","task":"Robot Manipulation","dataset_variant":"CALVIN","rows":19,"metrics":["avg. sequence length (D to D)"],"first_row_in_archive_order":{"model":"DreamVLA","paper":"/paper/dreamvla-a-vision-language-action-model-1","metrics":{"avg. sequence length (D to D)":"4.44"},"code_links":[{"title":"Zhangwenyao1/DreamVLA","url":"https://github.com/Zhangwenyao1/DreamVLA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-generalization-on-calvin","task":"Zero-shot Generalization","dataset_variant":"CALVIN","rows":5,"metrics":["Avg. sequence length"],"first_row_in_archive_order":{"model":"GR-MG","paper":"/paper/gr-mg-leveraging-partially-annotated-data-via","metrics":{"Avg. sequence length":"4.04"},"code_links":[{"title":"bytedance/GR-MG","url":"https://github.com/bytedance/GR-MG"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/dreamvla-a-vision-language-action-model-1","title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","date":"2025-07-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/univla-learning-to-act-anywhere-with-task","title":"UniVLA: Learning to Act Anywhere with Task-centric Latent Actions","date":"2025-05-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openhelix-a-short-survey-empirical-analysis","title":"OpenHelix: A Short Survey, Empirical Analysis, and Open-Source Dual-System VLA Model for Robotic Manipulation","date":"2025-05-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/up-vla-a-unified-understanding-and-prediction","title":"UP-VLA: A Unified Understanding and Prediction Model for Embodied Agent","date":"2025-01-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/video-prediction-policy-a-generalist-robot","title":"Video Prediction Policy: A Generalist Robot Policy with Predictive Visual Representations","date":"2024-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-generalist-robot-policies-what","title":"Towards Generalist Robot Policies: What Matters in Building Vision-Language-Action Models","date":"2024-12-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-diffusion-transformer-policies-with","title":"Efficient Diffusion Transformer Policies with Mixture of Expert Denoisers for Multitask Learning","date":"2024-12-17","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vidman-exploiting-implicit-dynamics-from","title":"VidMan: Exploiting Implicit Dynamics from Video Diffusion Model for Effective Robot Manipulation","date":"2024-11-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/towards-synergistic-generalized-and-efficient","title":"Towards Synergistic, Generalized, and Efficient Dual-System for Robotic Manipulation","date":"2024-10-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gr-mg-leveraging-partially-annotated-data-via","title":"GR-MG: Leveraging Partially Annotated Data via Multi-Modal Goal-Conditioned Policy","date":"2024-08-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/robouniview-visual-language-model-with","title":"RoboUniView: Visual-Language Model with Unified View Representation for Robotic Manipulation","date":"2024-06-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openvla-an-open-source-vision-language-action","title":"OpenVLA: An Open-Source Vision-Language-Action Model","date":"2024-06-13","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/from-llms-to-actions-latent-codes-as-bridges","title":"From LLMs to Actions: Latent Codes as Bridges in Hierarchical Robot Control","date":"2024-05-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/3d-diffuser-actor-policy-diffusion-with-3d","title":"3D Diffuser Actor: Policy Diffusion with 3D Scene Representations","date":"2024-02-18","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/unleashing-large-scale-video-generative-pre","title":"Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation","date":"2023-12-20","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-language-foundation-models-as","title":"Vision-Language Foundation Models as Effective Robot Imitators","date":"2023-11-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-universal-policies-via-text-guided","title":"Learning Universal Policies via Text-Guided Video Generation","date":"2023-01-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/rt-1-robotics-transformer-for-real-world","title":"RT-1: Robotics Transformer for Real-World Control at Scale","date":"2022-12-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":11,"samples_harvested":70,"samples_ran":24,"samples_unverified":46,"pointer_only_for_licence":4,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}