{"url":"/dataset/pku-mmd","name":"PKU-MMD","full_name":"PKU-MMD","description_markdown":"The **PKU-MMD** dataset is a large skeleton-based action detection dataset. It contains 1076 long untrimmed video sequences performed by 66 subjects in three camera views. 51 action categories are annotated, resulting almost 20,000 action instances and 5.4 million frames in total. Similar to NTU RGB+D, there are also two recommended evaluate protocols, i.e. cross-subject and cross-view.\r\n\r\nSource: [Co-occurrence Feature Learning from Skeleton Data for Action Recognition and Detection with Hierarchical Aggregation](https://arxiv.org/abs/1804.06055)\r\nImage Source: [https://www.icst.pku.edu.cn/struct/Projects/PKUMMD.html](https://www.icst.pku.edu.cn/struct/Projects/PKUMMD.html)","description_withheld":null,"homepage":"https://www.icst.pku.edu.cn/struct/Projects/PKUMMD.html","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/pku-mmd-a-large-scale-benchmark-for","title":"PKU-MMD: A Large Scale Benchmark for Continuous Multi-Modal Human Action Understanding","first_author":"Chunhui Liu","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Skeleton Based Action Recognition","url":"/task/skeleton-based-action-recognition","datasets_with_task":"/datasets/task/skeleton-based-action-recognition"},{"name":"Action Recognition In Videos","url":"/task/action-recognition-in-videos-2","datasets_with_task":"/datasets/task/action-recognition-in-videos-2"},{"name":"Zero Shot Skeletal Action Recognition","url":"/task/zero-shot-skeletal-action-recognition","datasets_with_task":"/datasets/task/zero-shot-skeletal-action-recognition"},{"name":"Generalized Zero Shot skeletal action recognition","url":"/task/generalized-zero-shot-skeletal-action","datasets_with_task":"/datasets/task/generalized-zero-shot-skeletal-action"},{"name":"Unsupervised Skeleton Based Action Recognition","url":"/task/unsupervised-skeleton-based-action","datasets_with_task":"/datasets/task/unsupervised-skeleton-based-action"}],"languages":[],"variants":["PKU-MMD"],"data_loaders":[],"num_papers_in_archive":72,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-skeletal-action-recognition-on-pku","task":"Zero Shot Skeletal Action Recognition","dataset_variant":"PKU-MMD","rows":7,"metrics":["Random Split Accuracy"],"first_row_in_archive_order":{"model":"PGFA","paper":"/paper/zero-shot-skeleton-based-action-recognition-2","metrics":{"Random Split Accuracy":"87.80"},"code_links":[{"title":"kaai520/PGFA","url":"https://github.com/kaai520/PGFA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-pku-mmd","task":"Action Recognition In Videos","dataset_variant":"PKU-MMD","rows":5,"metrics":["X-Sub","X-View"],"first_row_in_archive_order":{"model":"DSCNet (RGB + Pose)","paper":"/paper/a-dense-sparse-complementary-network-for","metrics":{"X-Sub":"97.4","X-View":"98.8"},"code_links":[{"title":"Maxchengqin/DSCNet","url":"https://github.com/Maxchengqin/DSCNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/skeleton-based-action-recognition-on-pku-mmd","task":"Skeleton Based Action Recognition","dataset_variant":"PKU-MMD","rows":4,"metrics":["mAP@0.50 (CV)","mAP@0.50 (CS)","Accuracy (Cross-Subject)"],"first_row_in_archive_order":{"model":"RF-Action","paper":"/paper/making-the-invisible-visible-action","metrics":{"mAP@0.50 (CS)":"92.9","mAP@0.50 (CV)":"94.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/generalized-zero-shot-skeletal-action-2","task":"Generalized Zero Shot skeletal action recognition","dataset_variant":"PKU-MMD","rows":1,"metrics":["Random Split Harmonic Mean"],"first_row_in_archive_order":{"model":"SA-DVAE","paper":"/paper/sa-dvae-improving-zero-shot-skeleton-based","metrics":{"Random Split Harmonic Mean":"54.72"},"code_links":[{"title":"pha123661/SA-DVAE","url":"https://github.com/pha123661/SA-DVAE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/zero-shot-skeleton-based-action-recognition-2","title":"Zero-shot Skeleton-based Action Recognition with Prototype-guided Feature Alignment","date":"2025-07-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tdsm-triplet-diffusion-for-skeleton-text","title":"TDSM: Triplet Diffusion for Skeleton-Text Matching in Zero-Shot Action Recognition","date":"2024-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/epam-net-an-efficient-pose-driven-attention","title":"EPAM-Net: An Efficient Pose-driven Attention-guided Multimodal Network for Video Action Recognition","date":"2024-08-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sa-dvae-improving-zero-shot-skeleton-based","title":"SA-DVAE: Improving Zero-Shot Skeleton-Based Action Recognition by Disentangled Variational Autoencoders","date":"2024-07-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/fine-grained-side-information-guided-dual","title":"Fine-Grained Side Information Guided Dual-Prompts for Zero-Shot Skeleton Action Recognition","date":"2024-04-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-dense-sparse-complementary-network-for","title":"A Dense-Sparse Complementary Network for Human Action Recognition based on RGB and Skeleton Modalities","date":"2023-12-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dvanet-disentangling-view-and-action-features","title":"DVANet: Disentangling View and Action Features for Multi-View Action Recognition","date":"2023-12-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/zero-shot-skeleton-based-action-recognition","title":"Zero-shot Skeleton-based Action Recognition via Mutual Information Estimation and Maximization","date":"2023-08-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hyperbolic-self-paced-learning-for-self","title":"HYperbolic Self-Paced Learning for Self-Supervised Skeleton-based Action Representations","date":"2023-03-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":2,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mmnet-a-model-based-multimodal-network-for","title":"MMNet: A Model-Based Multimodal Network for Human Action Recognition in RGB-D Videos","date":"2022-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multimodal-fusion-via-teacher-student-network","title":"Multimodal Fusion via Teacher-Student Network for Indoor Action Recognition","date":"2021-05-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/syntactically-guided-generative-embeddings-1","title":"Syntactically Guided Generative Embeddings for Zero-Shot Skeleton Action Recognition","date":"2021-01-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/making-the-invisible-visible-action","title":"Making the Invisible Visible: Action Recognition Through Walls and Occlusions","date":"2019-09-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/generalized-zero-and-few-shot-learning-via","title":"Generalized Zero- and Few-Shot Learning via Aligned Variational Autoencoders","date":"2018-12-05","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/co-occurrence-feature-learning-from-skeleton","title":"Co-occurrence Feature Learning from Skeleton Data for Action Recognition and Detection with Hierarchical Aggregation","date":"2018-04-17","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/skeleton-based-action-recognition-with-2","title":"Skeleton-based Action Recognition with Convolutional Neural Networks","date":"2017-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":20,"samples_ran":8,"samples_unverified":12,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}