{"url":"/dataset/humanml3d","name":"HumanML3D","full_name":null,"description_markdown":"HumanML3D is a 3D human motion-language dataset that originates from a combination of HumanAct12 and Amass dataset. It covers a broad range of human actions such as daily activities (e.g., 'walking', 'jumping'), sports (e.g., 'swimming', 'playing golf'), acrobatics (e.g., 'cartwheel') and artistry (e.g., 'dancing'). Overall, HumanML3D dataset consists of 14,616 motions and 44,970 descriptions composed by 5,371 distinct words. The total length of motions amounts to 28.59 hours. The average motion length is 7.1 seconds, while average description length is 12 words.","description_withheld":null,"homepage":"https://github.com/EricGuo5513/HumanML3D","introduced_date":"2022-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/generating-diverse-and-natural-3d-human","title":"Generating Diverse and Natural 3D Human Motions From Text","first_author":"Chuan Guo","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"3D","url":"/datasets/modality/3d"}],"tasks":[{"name":"Motion Synthesis","url":"/task/motion-synthesis","datasets_with_task":"/datasets/task/motion-synthesis"},{"name":"Motion Captioning","url":"/task/motion-captioning","datasets_with_task":"/datasets/task/motion-captioning"},{"name":"Motion Generation","url":"/task/motion-generation","datasets_with_task":"/datasets/task/motion-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["HumanML3D"],"data_loaders":[],"num_papers_in_archive":201,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/motion-synthesis-on-humanml3d","task":"Motion Synthesis","dataset_variant":"HumanML3D","rows":37,"metrics":["FID","R Precision Top3","Diversity","Multimodality"],"first_row_in_archive_order":{"model":"Motion Anything","paper":"/paper/motion-anything-any-to-motion-generation","metrics":{"Diversity":"9.521","FID":"0.028","Multimodality":"2.705","R Precision Top3":"0.829"},"code_links":[{"title":"steve-zeyu-zhang/MotionAnything","url":"https://github.com/steve-zeyu-zhang/MotionAnything"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/motion-captioning-on-humanml3d","task":"Motion Captioning","dataset_variant":"HumanML3D","rows":4,"metrics":["BLEU-4","BERTScore"],"first_row_in_archive_order":{"model":"ST-MLP","paper":"/paper/guided-attention-for-interpretable-motion","metrics":{"BERTScore":"40.3","BLEU-4":"25.0"},"code_links":[{"title":"rd20karim/m2t-interpretable","url":"https://github.com/rd20karim/m2t-interpretable"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/motion-anything-any-to-motion-generation","title":"Motion Anything: Any to Motion Generation","date":"2025-03-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/motionpcm-real-time-motion-synthesis-with","title":"MotionPCM: Real-Time Motion Synthesis with Phased Consistency Model","date":"2025-01-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/free-t2m-frequency-enhanced-text-to-motion","title":"Free-T2M: Frequency Enhanced Text-to-Motion Diffusion Model With Consistency Loss","date":"2025-01-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/discord-discrete-tokens-to-continuous-motion","title":"DisCoRD: Discrete Tokens to Continuous Motion via Rectified Flow Decoding","date":"2024-11-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bipo-bidirectional-partial-occlusion-network-1","title":"BiPO: Bidirectional Partial Occlusion Network for Text-to-Motion Synthesis","date":"2024-11-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bad-bidirectional-auto-regressive-diffusion","title":"BAD: Bidirectional Auto-regressive Diffusion for Text-to-Motion Generation","date":"2024-09-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/motionlcm-real-time-controllable-motion","title":"MotionLCM: Real-time Controllable Motion Generation via Latent Consistency Model","date":"2024-04-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mcm-multi-condition-motion-synthesis-1","title":"MCM: Multi-condition Motion Synthesis Framework","date":"2024-04-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bamm-bidirectional-autoregressive-motion","title":"BAMM: Bidirectional Autoregressive Motion Model","date":"2024-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/parco-part-coordinating-text-to-motion","title":"ParCo: Part-Coordinating Text-to-Motion Synthesis","date":"2024-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/motion-mamba-efficient-and-long-sequence","title":"Motion Mamba: Efficient and Long Sequence Motion Generation","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/guess-gradually-enriching-synthesis-for-text","title":"GUESS:GradUally Enriching SyntheSis for Text-Driven Human Motion Generation","date":"2024-01-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/finemogen-fine-grained-spatio-temporal-motion-1","title":"FineMoGen: Fine-Grained Spatio-Temporal Motion Generation and Editing","date":"2023-12-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mmm-generative-masked-motion-model","title":"MMM: Generative Masked Motion Model","date":"2023-12-06","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/emdm-efficient-motion-diffusion-model-for","title":"EMDM: Efficient Motion Diffusion Model for Fast and High-Quality Motion Generation","date":"2023-12-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/momask-generative-masked-modeling-of-3d-human","title":"MoMask: Generative Masked Modeling of 3D Human Motions","date":"2023-11-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/motion2language-unsupervised-learning-of","title":"Motion2Language, unsupervised learning of synchronized semantic motion segmentation","date":"2023-10-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/guided-attention-for-interpretable-motion","title":"Guided Attention for Interpretable Motion Captioning","date":"2023-10-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/act-as-you-wish-fine-grained-control-of","title":"Act As You Wish: Fine-Grained Control of Motion Diffusion Model with Hierarchical Semantic Graphs","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fg-t2m-fine-grained-text-driven-human-motion","title":"Fg-T2M: Fine-Grained Text-Driven Human Motion Generation via Diffusion Model","date":"2023-09-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/diversemotion-towards-diverse-human-motion","title":"DiverseMotion: Towards Diverse Human Motion Generation via Discrete Diffusion","date":"2023-09-04","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/attt2m-text-driven-human-motion-generation-1","title":"AttT2M: Text-Driven Human Motion Generation with Multi-Perspective Attention Mechanism","date":"2023-09-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/motiongpt-human-motion-as-a-foreign-language","title":"MotionGPT: Human Motion as a Foreign Language","date":"2023-06-26","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/make-an-animation-large-scale-text","title":"Make-An-Animation: Large-Scale Text-conditional 3D Human Motion Generation","date":"2023-05-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tm2d-bimodality-driven-3d-dance-generation","title":"TM2D: Bimodality Driven 3D Dance Generation via Music-Text Integration","date":"2023-04-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/remodiffuse-retrieval-augmented-motion","title":"ReMoDiffuse: Retrieval-Augmented Motion Diffusion Model","date":"2023-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/t2m-gpt-generating-human-motion-from-textual","title":"T2M-GPT: Generating Human Motion from Textual Descriptions with Discrete Representations","date":"2023-01-15","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/executing-your-commands-via-motion-diffusion","title":"Executing your Commands via Motion Diffusion in Latent Space","date":"2022-12-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffusion-motion-generate-text-guided-3d","title":"Diffusion Motion: Generate Text-Guided 3D Human Motion by Diffusion Model","date":"2022-10-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/human-motion-diffusion-model","title":"Human Motion Diffusion Model","date":"2022-09-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/motiondiffuse-text-driven-human-motion","title":"MotionDiffuse: Text-Driven Human Motion Generation with Diffusion Model","date":"2022-08-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/tm2t-stochastic-and-tokenized-modeling-for","title":"TM2T: Stochastic and Tokenized Modeling for the Reciprocal Generation of 3D Human Motions and Texts","date":"2022-07-04","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/generating-diverse-and-natural-3d-human","title":"Generating Diverse and Natural 3D Human Motions From Text","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":12,"samples_harvested":50,"samples_ran":30,"samples_unverified":20,"pointer_only_for_licence":15,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}