{"url":"/dataset/jhmdb","name":"JHMDB","full_name":"Joint-annotated Human Motion Data Base","description_markdown":"**JHMDB** is an action recognition dataset that consists of 960 video sequences belonging to 21 actions. It is a subset of the larger HMDB51 dataset collected from digitized movies and YouTube videos. The dataset contains video and annotation for puppet flow per frame (approximated optimal flow on the person), puppet mask per frame, joint positions per frame, action label per clip and meta label per clip (camera motion, visible body parts, camera viewpoint, number of people, video quality).\r\n\r\nSource: [Unsupervised Deep Metric Learning via Orthogonality based Probabilistic Loss](https://arxiv.org/abs/2008.09880)\r\nImage Source: [https://arxiv.org/pdf/1712.06316.pdf](https://arxiv.org/pdf/1712.06316.pdf)","description_withheld":null,"homepage":"http://jhmdb.is.tue.mpg.de/","introduced_date":"2013-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"Towards Understanding Action Recognition","first_author":null,"url":"https://doi.org/10.1109/ICCV.2013.396"},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Pose Estimation","url":"/task/pose-estimation","datasets_with_task":"/datasets/task/pose-estimation"},{"name":"Action Detection","url":"/task/action-detection","datasets_with_task":"/datasets/task/action-detection"},{"name":"2D Human Pose Estimation","url":"/task/2d-human-pose-estimation","datasets_with_task":"/datasets/task/2d-human-pose-estimation"},{"name":"Skeleton Based Action Recognition","url":"/task/skeleton-based-action-recognition","datasets_with_task":"/datasets/task/skeleton-based-action-recognition"},{"name":"Referring Expression Segmentation","url":"/task/referring-expression-segmentation","datasets_with_task":"/datasets/task/referring-expression-segmentation"},{"name":"Open Vocabulary Action Detection","url":"/task/open-vocabulary-action-detection","datasets_with_task":"/datasets/task/open-vocabulary-action-detection"}],"languages":[],"variants":["JHMDB (2D poses only)","J-HMDB","J-HMBD Early Action","JHMDB Pose Tracking","J-HMDB-21","JHMDB"],"data_loaders":[{"repo":"https://github.com/open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose","frameworks":[]},{"repo":"https://github.com/open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2/blob/master/tools/data/jhmdb/README.md","frameworks":["pytorch"]}],"num_papers_in_archive":249,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/referring-expression-segmentation-on-j-hmdb","task":"Referring Expression Segmentation","dataset_variant":"J-HMDB","rows":21,"metrics":["AP","IoU overall","IoU mean","Precision@0.5","Precision@0.6","Precision@0.7","Precision@0.8","Precision@0.9"],"first_row_in_archive_order":{"model":"SgMg (Video-Swin-B)","paper":"/paper/spectrum-guided-multi-granularity-referring","metrics":{"AP":"0.450","IoU mean":"0.725","IoU overall":"0.737","Precision@0.5":"0.972","Precision@0.6":"0.917","Precision@0.7":"0.714","Precision@0.8":"0.225","Precision@0.9":"0.003"},"code_links":[{"title":"bo-miao/sgmg","url":"https://github.com/bo-miao/sgmg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-detection-on-j-hmdb","task":"Action Detection","dataset_variant":"J-HMDB","rows":18,"metrics":["Frame-mAP 0.5","Video-mAP 0.2","Video-mAP 0.5"],"first_row_in_archive_order":{"model":"SiA","paper":"/paper/scaling-open-vocabulary-action-detection","metrics":{"Frame-mAP 0.5":"88.5"},"code_links":[{"title":"siatheindochinese/sia_act_placeholder","url":"https://github.com/siatheindochinese/sia_act_placeholder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/skeleton-based-action-recognition-on-j-hmdb","task":"Skeleton Based Action Recognition","dataset_variant":"J-HMDB","rows":13,"metrics":["Accuracy (RGB+pose)","Accuracy (pose)"],"first_row_in_archive_order":{"model":"Potion","paper":"/paper/potion-pose-motion-representation-for-action","metrics":{"Accuracy (RGB+pose)":"90.4","Accuracy (pose)":"67.9"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/skeleton-based-action-recognition-on-jhmdb-2d","task":"Skeleton Based Action Recognition","dataset_variant":"JHMDB (2D poses only)","rows":6,"metrics":["Average accuracy of 3 splits","Accuracy","No. parameters"],"first_row_in_archive_order":{"model":"HT-ConvNet","paper":"/paper/hierarchical-temporal-convolution-network","metrics":{"Accuracy":"86.1","Average accuracy of 3 splits":"86.1","No. parameters":"1.75"},"code_links":[{"title":"Gbouna/HT-ConvNet","url":"https://github.com/Gbouna/HT-ConvNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/2d-human-pose-estimation-on-jhmdb-2d-poses","task":"2D Human Pose Estimation","dataset_variant":"JHMDB (2D poses only)","rows":5,"metrics":["PCK"],"first_row_in_archive_order":{"model":"DeciWatch","paper":"/paper/deciwatch-a-simple-baseline-for-10x-efficient","metrics":{"PCK":"98.8"},"code_links":[{"title":"cure-lab/DeciWatch","url":"https://github.com/cure-lab/DeciWatch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-estimation-on-j-hmdb","task":"Pose Estimation","dataset_variant":"J-HMDB","rows":5,"metrics":["Mean PCK@0.2","Mean PCK@0.1","Mean PCK@0.05"],"first_row_in_archive_order":{"model":"SimpleBaseline + HANet","paper":"/paper/kinematic-aware-hierarchical-attention","metrics":{"Mean PCK@0.05":"91.9","Mean PCK@0.1":"98.3","Mean PCK@0.2":"99.6"},"code_links":[{"title":"kyungminjin/hanet","url":"https://github.com/kyungminjin/hanet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/skeleton-based-action-recognition-on-jhmdb","task":"Skeleton Based Action Recognition","dataset_variant":"JHMDB Pose Tracking","rows":3,"metrics":["PCK@0.1","PCK@0.2","PCK@0.3","PCK@0.4","PCK@0.5"],"first_row_in_archive_order":{"model":"mgPFF+ft 1st","paper":"/paper/multigrid-predictive-filter-flow-for","metrics":{"PCK@0.1":"58.4","PCK@0.2":"78.1","PCK@0.3":"85.9","PCK@0.4":"89.8","PCK@0.5":"92.4"},"code_links":[{"title":"aimerykong/predictive-filter-flow","url":"https://github.com/aimerykong/predictive-filter-flow"},{"title":"bestaar/predictiveFilterFlow","url":"https://github.com/bestaar/predictiveFilterFlow"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/skeleton-based-action-recognition-on-j-hmbd","task":"Skeleton Based Action Recognition","dataset_variant":"J-HMBD Early Action","rows":2,"metrics":["10%"],"first_row_in_archive_order":{"model":"DR^2N","paper":"/paper/relational-autoencoder-for-feature-extraction","metrics":{"10%":"60.6"},"code_links":[{"title":"ser-art/RAE-vs-AE","url":"https://github.com/ser-art/RAE-vs-AE"},{"title":"rk68657/AutoEncoders","url":"https://github.com/rk68657/AutoEncoders"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-action-detection-on-jhmdb","task":"Open Vocabulary Action Detection","dataset_variant":"JHMDB","rows":1,"metrics":["val mAP"],"first_row_in_archive_order":{"model":"SiA","paper":"/paper/scaling-open-vocabulary-action-detection","metrics":{"val mAP":"57.1"},"code_links":[{"title":"siatheindochinese/sia_act_placeholder","url":"https://github.com/siatheindochinese/sia_act_placeholder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/scaling-open-vocabulary-action-detection","title":"Scaling Open-Vocabulary Action Detection","date":"2025-04-04","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/poseidon-a-vit-based-architecture-for-multi","title":"Poseidon: A ViT-based Architecture for Multi-Frame Pose Estimation with Adaptive Frame Weighting and Multi-Scale Feature Fusion","date":"2025-01-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-temporal-convolution-network","title":"Hierarchical Temporal Convolution Network:Towards Privacy-Centric Activity Recognition","date":"2024-12-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/spectrum-guided-multi-granularity-referring","title":"Spectrum-guided Multi-granularity Referring Video Object Segmentation","date":"2023-07-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/soc-semantic-assisted-object-cluster-for","title":"SOC: Semantic-Assisted Object Cluster for Referring Video Object Segmentation","date":"2023-05-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/kinematic-aware-hierarchical-attention","title":"Kinematic-aware Hierarchical Attention Network for Human Pose Estimation in Videos","date":"2022-11-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/holistic-interaction-transformer-network-for","title":"Holistic Interaction Transformer Network for Action Detection","date":"2022-10-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deeply-interleaved-two-stream-encoder-for","title":"Deeply Interleaved Two-Stream Encoder for Referring Video Segmentation","date":"2022-03-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deciwatch-a-simple-baseline-for-10x-efficient","title":"DeciWatch: A Simple Baseline for 10x Efficient 2D and 3D Pose Estimation","date":"2022-03-16","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-referring-video-object","title":"End-to-End Referring Video Object Segmentation with Multimodal Transformers","date":"2021-11-29","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hierarchical-interaction-network-for-video","title":"Hierarchical interaction network for video object segmentation from referring expressions","date":"2021-11-22","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/do-different-tracking-tasks-require-different","title":"Do Different Tracking Tasks Require Different Appearance Models?","date":"2021-07-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":6,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-progressive-comprehension-for","title":"Cross-Modal Progressive Comprehension for Referring Segmentation","date":"2021-05-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/collaborative-spatial-temporal-modeling-for","title":"Collaborative Spatial-Temporal Modeling for Language-Queried Video Actor Segmentation","date":"2021-05-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/clawcranenet-leveraging-object-level-relation","title":"ClawCraneNet: Leveraging Object-level Relation for Text-based Video Segmentation","date":"2021-03-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/referring-segmentation-in-images-and-videos","title":"Referring Segmentation in Images and Videos with Cross-Modal Self-Attention Network","date":"2021-02-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/actor-and-action-modular-network-for-text","title":"Actor and Action Modular Network for Text-based Video Segmentation","date":"2020-11-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pose-and-joint-aware-action-recognition","title":"Pose And Joint-Aware Action Recognition","date":"2020-10-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/finding-action-tubes-with-a-sparse-to-dense","title":"Finding Action Tubes with a Sparse-to-Dense Framework","date":"2020-08-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/polar-relative-positional-encoding-for-video","title":"Polar Relative Positional Encoding for Video-Language Segmentation","date":"2020-07-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/visual-textual-capsule-routing-for-text-based","title":"Visual-Textual Capsule Routing for Text-Based Video Segmentation","date":"2020-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/context-modulated-dynamic-networks-for-actor","title":"Context Modulated Dynamic Networks for Actor and Action Video Segmentation with Language Queries","date":"2020-04-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/actions-as-moving-points","title":"Actions as Moving Points","date":"2020-01-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/you-only-watch-once-a-unified-cnn","title":"You Only Watch Once: A Unified CNN Architecture for Real-Time Spatiotemporal Action Localization","date":"2019-11-15","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":3,"samples_unverified":9,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hierarchical-self-attention-network-for","title":"Hierarchical Self-Attention Network for Action Localization in Videos","date":"2019-10-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/asymmetric-cross-guided-attention-network-for","title":"Asymmetric Cross-Guided Attention Network for Actor and Action Video Segmentation From Natural Language Query","date":"2019-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dynamic-kernel-distillation-for-efficient","title":"Dynamic Kernel Distillation for Efficient Pose Estimation in Videos","date":"2019-08-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/make-skeleton-based-action-recognition-model-1","title":"Make Skeleton-based Action Recognition Model Smaller, Faster and Better","date":"2019-07-23","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/pa3d-pose-action-3d-machine-for-video","title":"PA3D: Pose-Action 3D Machine for Video Recognition","date":"2019-06-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/tacnet-transition-aware-context-network-for-1","title":"TACNet: Transition-Aware Context Network for Spatio-Temporal Action Detection","date":"2019-05-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/simple-yet-efficient-real-time-pose-based","title":"Simple yet efficient real-time pose-based action recognition","date":"2019-04-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multigrid-predictive-filter-flow-for","title":"Multigrid Predictive Filter Flow for Unsupervised Learning on Videos","date":"2019-04-02","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dance-with-flow-two-in-one-stream-action","title":"Dance with Flow: Two-in-One Stream Action Detection","date":"2019-04-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/star-net-action-recognition-using-spatio","title":"STAR-Net: Action Recognition using Spatio-Temporal Activation Reprojection","date":"2019-02-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tracking-emerges-by-colorizing-videos","title":"Tracking Emerges by Colorizing Videos","date":"2018-06-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/potion-pose-motion-representation-for-action","title":"PoTion: Pose MoTion Representation for Action Recognition","date":"2018-06-01","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/simple-baselines-for-human-pose-estimation","title":"Simple Baselines for Human Pose Estimation and Tracking","date":"2018-04-17","rows_on_this_dataset":1,"code_links":27,"syntology":null},{"paper":"/paper/actor-and-action-video-segmentation-from-a","title":"Actor and Action Video Segmentation from a Sentence","date":"2018-03-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/relational-autoencoder-for-feature-extraction","title":"Relational Autoencoder for Feature Extraction","date":"2018-02-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/lstm-pose-machines","title":"LSTM Pose Machines","date":"2017-12-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/graph-attention-networks","title":"Graph Attention Networks","date":"2017-10-30","rows_on_this_dataset":1,"code_links":93,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":106,"samples_ran":50,"samples_unverified":56,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rpan-an-end-to-end-recurrent-pose-attention-1","title":"RPAN: An End-to-End Recurrent Pose-Attention Network for Action Recognition in Videos","date":"2017-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tracking-by-natural-language-specification","title":"Tracking by Natural Language Specification","date":"2017-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ava-a-video-dataset-of-spatio-temporally","title":"AVA: A Video Dataset of Spatio-temporally Localized Atomic Visual Actions","date":"2017-05-23","rows_on_this_dataset":1,"code_links":9,"syntology":null},{"paper":"/paper/quo-vadis-action-recognition-a-new-model-and","title":"Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset","date":"2017-05-22","rows_on_this_dataset":1,"code_links":34,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":28,"samples_ran":16,"samples_unverified":12,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chained-multi-stream-networks-exploiting-pose","title":"Chained Multi-stream Networks Exploiting Pose, Motion, and Appearance for Action Classification and Detection","date":"2017-04-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/tube-convolutional-neural-network-t-cnn-for","title":"Tube Convolutional Neural Network (T-CNN) for Action Detection in Videos","date":"2017-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/flownet-20-evolution-of-optical-flow","title":"FlowNet 2.0: Evolution of Optical Flow Estimation with Deep Networks","date":"2016-12-06","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":2,"samples_unverified":19,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-region-two-stream-r-cnn-for-action","title":"Multi-region two-stream R-CNN for action detection","date":"2016-09-17","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/actionness-estimation-using-hybrid-fully","title":"Actionness Estimation Using Hybrid Fully Convolutional Networks","date":"2016-04-25","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/segmentation-from-natural-language","title":"Segmentation from Natural Language Expressions","date":"2016-03-20","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/convolutional-pose-machines","title":"Convolutional Pose Machines","date":"2016-01-30","rows_on_this_dataset":1,"code_links":50,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/finding-action-tubes","title":"Finding Action Tubes","date":"2014-11-21","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":11,"samples_harvested":230,"samples_ran":98,"samples_unverified":132,"pointer_only_for_licence":72,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}