{"url":"/dataset/hmdb51","name":"HMDB51","full_name":null,"description_markdown":"The **HMDB51** dataset is a large collection of realistic videos from various sources, including movies and web videos. The dataset is composed of 6,766 video clips from 51 action categories (such as “jump”, “kiss” and “laugh”), with each category containing at least 101 clips. The original evaluation scheme uses three different training/testing splits. In each split, each action class has 70 clips for training and 30 clips for testing. The average accuracy over these three splits is used to measure the final performance.\r\n\r\nSource: [Action Recognition with Trajectory-Pooled Deep-Convolutional Descriptors](https://arxiv.org/abs/1505.04868)\r\nImage Source: [https://serre-lab.clps.brown.edu/resource/hmdb-a-large-human-motion-database](https://serre-lab.clps.brown.edu/resource/hmdb-a-large-human-motion-database)","description_withheld":null,"homepage":"https://serre-lab.clps.brown.edu/resource/hmdb-a-large-human-motion-database","introduced_date":"2011-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"HMDB: A large video database for human motion recognition","first_author":null,"url":"https://doi.org/10.1109/ICCV.2011.6126543"},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Skeleton Based Action Recognition","url":"/task/skeleton-based-action-recognition","datasets_with_task":"/datasets/task/skeleton-based-action-recognition"},{"name":"Action Classification","url":"/task/action-classification","datasets_with_task":"/datasets/task/action-classification"},{"name":"Action Recognition In Videos","url":"/task/action-recognition-in-videos-2","datasets_with_task":"/datasets/task/action-recognition-in-videos-2"},{"name":"Human Activity Recognition","url":"/task/human-activity-recognition","datasets_with_task":"/datasets/task/human-activity-recognition"},{"name":"Zero-Shot Action Recognition","url":"/task/zero-shot-action-recognition","datasets_with_task":"/datasets/task/zero-shot-action-recognition"},{"name":"Self-Supervised Action Recognition","url":"/task/self-supervised-action-recognition","datasets_with_task":"/datasets/task/self-supervised-action-recognition"},{"name":"Few Shot Action Recognition","url":"/task/few-shot-action-recognition","datasets_with_task":"/datasets/task/few-shot-action-recognition"},{"name":"Self-Supervised Action Recognition Linear","url":"/task/self-supervised-action-recognition-linear","datasets_with_task":"/datasets/task/self-supervised-action-recognition-linear"},{"name":"Self-supervised Video Retrieval","url":"/task/self-supervised-video-retrieval","datasets_with_task":"/datasets/task/self-supervised-video-retrieval"}],"languages":[],"variants":["HMDB51-skeleton","HMDB51 (finetuned)","HMDB51","HMDB-51"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/jili5044/hmdb51","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/pytorch/vision","url":"https://pytorch.org/vision/stable/generated/torchvision.datasets.HMDB51.html","frameworks":["pytorch"]},{"repo":"https://github.com/voxel51/fiftyone","url":"https://docs.voxel51.com/user_guide/dataset_zoo/datasets.html#hmbd51","frameworks":["tf","pytorch"]},{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/hmdb51-dataset","frameworks":["tf","pytorch"]},{"repo":"https://github.com/open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2/blob/master/tools/data/hmdb51/README.md","frameworks":["pytorch"]}],"num_papers_in_archive":839,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-recognition-in-videos-on-hmdb-51","task":"Action Recognition","dataset_variant":"HMDB-51","rows":77,"metrics":["Average accuracy of 3 splits"],"first_row_in_archive_order":{"model":"VideoMAE V2-g","paper":"/paper/videomae-v2-scaling-video-masked-autoencoders","metrics":{"Average accuracy of 3 splits":"88.7"},"code_links":[{"title":"OpenGVLab/VideoMAEv2","url":"https://github.com/OpenGVLab/VideoMAEv2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/self-supervised-action-recognition-on-hmdb51","task":"Self-Supervised Action Recognition","dataset_variant":"HMDB51","rows":48,"metrics":["Top-1 Accuracy","Pre-Training Dataset","Frozen"],"first_row_in_archive_order":{"model":"MVD (ViT-B)","paper":"/paper/masked-video-distillation-rethinking-masked","metrics":{"Frozen":"false","Pre-Training Dataset":"Kinetics400","Top-1 Accuracy":"79.7"},"code_links":[{"title":"ruiwang2021/mvd","url":"https://github.com/ruiwang2021/mvd"},{"title":"2023-MindSpore-4/Code-5","url":"https://github.com/2023-MindSpore-4/Code-5/tree/main/MVD"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/MVD"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/MVD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-action-recognition-on-hmdb51","task":"Zero-Shot Action Recognition","dataset_variant":"HMDB51","rows":29,"metrics":["Top-1 Accuracy","Top-5 Accuracy","Accuracy"],"first_row_in_archive_order":{"model":"MOV (ViT-L/14)","paper":"/paper/multimodal-open-vocabulary-video","metrics":{"Top-1 Accuracy":"64.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/self-supervised-action-recognition-on-hmdb51-1","task":"Self-Supervised Action Recognition","dataset_variant":"HMDB51 (finetuned)","rows":14,"metrics":["Top-1 Accuracy","Pretraining Dataset"],"first_row_in_archive_order":{"model":"BraVe:V-FA (TSM-50x2)","paper":"/paper/broaden-your-views-for-self-supervised-video","metrics":{"Top-1 Accuracy":"77.8"},"code_links":[{"title":"deepmind/brave","url":"https://github.com/deepmind/brave"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-action-recognition-on-hmdb51","task":"Few Shot Action Recognition","dataset_variant":"HMDB51","rows":7,"metrics":["1:1 Accuracy"],"first_row_in_archive_order":{"model":"STRM","paper":"/paper/spatio-temporal-relation-modeling-for-few","metrics":{"1:1 Accuracy":"77.3"},"code_links":[{"title":"Anirudh257/strm","url":"https://github.com/Anirudh257/strm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/skeleton-based-action-recognition-on-hmdb51","task":"Skeleton Based Action Recognition","dataset_variant":"HMDB51","rows":2,"metrics":["Accuracy","Average accuracy of 3 splits"],"first_row_in_archive_order":{"model":"Structured Keypoint Pooling","paper":"/paper/unified-keypoint-based-action-recognition","metrics":{"Accuracy":"70.9"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-classification-on-hmdb51","task":"Action Classification","dataset_variant":"HMDB51","rows":1,"metrics":["Acc@1"],"first_row_in_archive_order":{"model":"DualPath w/ ViT-B/16 MLPs.","paper":"/paper/dual-path-adaptation-from-image-to-video","metrics":{"Acc@1":"75.6"},"code_links":[{"title":"park-jungin/dualpath","url":"https://github.com/park-jungin/dualpath"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-hmdb-51-1","task":"Action Recognition In Videos","dataset_variant":"HMDB-51","rows":1,"metrics":["Average accuracy of 3 splits"],"first_row_in_archive_order":{"model":"STM (ImageNet+Kinetics pretrain)","paper":"/paper/stm-spatiotemporal-and-motion-encoding-for","metrics":{"Average accuracy of 3 splits":"72.2"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-hmdb51","task":"Action Recognition","dataset_variant":"HMDB51","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"MSQNet","paper":"/paper/msqnet-actor-agnostic-action-recognition-with","metrics":{"Accuracy":"93.25"},"code_links":[{"title":"mondalanindya/msqnet","url":"https://github.com/mondalanindya/msqnet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/human-activity-recognition-on-hmdb51","task":"Human Activity Recognition","dataset_variant":"HMDB51","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Label-Ranker","paper":"/paper/label-ranker-self-aware-preference-for","metrics":{"Accuracy":"61.18%"},"code_links":[{"title":"Peihao-Xiang/Label-Ranker","url":"https://github.com/Peihao-Xiang/Label-Ranker"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/label-ranker-self-aware-preference-for","title":"Label Ranker: Self-Aware Preference for Classification Label Position in Visual Masked Self-Supervised Pre-Trained Model","date":"2025-03-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/iot-based-real-time-medical-related-human","title":"IoT-Based Real-Time Medical-Related Human Activity Recognition Using Skeletons and Multi-Stage Deep Learning for Healthcare","date":"2025-01-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dejavid-encoder-agnostic-learned-temporal","title":"DejaVid: Encoder-Agnostic Learned Temporal Matching for Video Classification","date":"2025-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/locate-gat-modeling-multi-scale-local-context","title":"LoCATe-GAT: Modeling Multi-Scale Local Context and Action Relationships for Zero-Shot Action Recognition","date":"2024-11-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/leveraging-temporal-contextualization-for","title":"Leveraging Temporal Contextualization for Video Action Recognition","date":"2024-04-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ost-refining-text-knowledge-with-optimal","title":"OST: Refining Text Knowledge with Optimal Spatio-Temporal Descriptor for General Video Recognition","date":"2023-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/asymmetric-masked-distillation-for-pre","title":"Asymmetric Masked Distillation for Pre-Training Small Foundation Models","date":"2023-11-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/zeroi2v-zero-cost-adaptation-of-pre-trained","title":"ZeroI2V: Zero-Cost Adaptation of Pre-trained Transformers from Image to Video","date":"2023-10-02","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/orthogonal-temporal-interpolation-for-zero","title":"Orthogonal Temporal Interpolation for Zero-Shot Video Recognition","date":"2023-08-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/msqnet-actor-agnostic-action-recognition-with","title":"Actor-agnostic Multi-label Action Recognition with Multi-modal Query","date":"2023-07-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/alternating-gradient-descent-and-mixture-of","title":"Alternating Gradient Descent and Mixture-of-Experts for Integrated Multimodal Perception","date":"2023-05-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/synthetic-sample-selection-for-generalized","title":"Synthetic Sample Selection for Generalized Zero-Shot Learning","date":"2023-04-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/victr-video-conditioned-text-representations","title":"VicTR: Video-conditioned Text Representations for Activity Recognition","date":"2023-04-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/videomae-v2-scaling-video-masked-autoencoders","title":"VideoMAE V2: Scaling Video Masked Autoencoders with Dual Masking","date":"2023-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unified-keypoint-based-action-recognition","title":"Unified Keypoint-based Action Recognition Framework via Structured Keypoint Pooling","date":"2023-03-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dual-path-adaptation-from-image-to-video","title":"Dual-path Adaptation from Image to Video Transformers","date":"2023-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/match-expand-and-improve-unsupervised","title":"MAtch, eXpand and Improve: Unsupervised Finetuning for Zero-Shot Action Recognition with Language Knowledge","date":"2023-03-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bidirectional-cross-modal-knowledge","title":"Bidirectional Cross-Modal Knowledge Exploration for Video Recognition with Pre-trained Vision-Language Models","date":"2022-12-31","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/similarity-contrastive-estimation-for-image","title":"Similarity Contrastive Estimation for Image and Video Soft Contrastive Self-Supervised Learning","date":"2022-12-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/masked-video-distillation-rethinking-masked","title":"Masked Video Distillation: Rethinking Masked Feature Modeling for Self-supervised Video Representation Learning","date":"2022-12-08","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/xkd-cross-modal-knowledge-distillation-with","title":"XKD: Cross-modal Knowledge Distillation with Domain Alignment for Video Representation Learning","date":"2022-11-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/efficient-video-representation-learning-via","title":"EVEREST: Efficient Masked Video Autoencoder by Removing Redundant Spatiotemporal Tokens","date":"2022-11-19","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/m-3-video-masked-motion-modeling-for-self","title":"Masked Motion Encoding for Self-Supervised Video Representation Learning","date":"2022-10-12","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/expanding-language-image-pretrained-models","title":"Expanding Language-Image Pretrained Models for General Video Recognition","date":"2022-08-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-open-vocabulary-video","title":"Multimodal Open-Vocabulary Video Classification via Pre-Trained Vision and Language Models","date":"2022-07-15","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/transferring-textual-knowledge-for-visual","title":"Revisiting Classifier: Transferring Vision-Language Models for Video Recognition","date":"2022-07-04","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/slic-self-supervised-learning-with-iterative-1","title":"SLIC: Self-Supervised Learning with Iterative Clustering for Human Action Videos","date":"2022-06-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learn2augment-learning-to-composite-videos","title":"Learn2Augment: Learning to Composite Videos for Data Augmentation in Action Recognition","date":"2022-06-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cross-modal-representation-learning-for-zero","title":"Cross-modal Representation Learning for Zero-shot Action Recognition","date":"2022-05-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hybrid-relation-guided-set-matching-for-few","title":"Hybrid Relation Guided Set Matching for Few-shot Action Recognition","date":"2022-04-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/alignment-uniformity-aware-representation","title":"Alignment-Uniformity aware Representation Learning for Zero-shot Video Classification","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-zero-shot-action-recognition","title":"Rethinking Zero-shot Action Recognition: Learning from Latent Atomic Actions","date":"2022-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/videomae-masked-autoencoders-are-data-1","title":"VideoMAE: Masked Autoencoders are Data-Efficient Learners for Self-Supervised Video Pre-Training","date":"2022-03-23","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":9,"samples_unverified":4,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spatio-temporal-relation-modeling-for-few","title":"Spatio-temporal Relation Modeling for Few-shot Action Recognition","date":"2021-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-video-transformer","title":"Self-supervised Video Transformer","date":"2021-12-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-audio-visual-representation","title":"Self-Supervised Audio-Visual Representation Learning with Relaxed Cross-Modal Synchronicity","date":"2021-11-09","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/high-order-tensor-pooling-with-attention-for","title":"High-order Tensor Pooling with Attention for Action Recognition","date":"2021-10-11","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-representation-learning-8","title":"Self-Supervised Video Representation Learning with Meta-Contrastive Network","date":"2021-08-19","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/elaborative-rehearsal-for-zero-shot-action","title":"Elaborative Rehearsal for Zero-shot Action Recognition","date":"2021-08-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vimpac-video-pre-training-via-masked-token","title":"VIMPAC: Video Pre-Training via Masked Token Prediction and Contrastive Learning","date":"2021-06-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-video-representation-learning-7","title":"Self-supervised Video Representation Learning with Cross-Stream Prototypical Contrasting","date":"2021-06-18","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/a-large-scale-study-on-unsupervised","title":"A Large-Scale Study on Unsupervised Spatiotemporal Representation Learning","date":"2021-04-29","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/vidtr-video-transformer-without-convolutions","title":"VidTr: Video Transformer Without Convolutions","date":"2021-04-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/broaden-your-views-for-self-supervised-video","title":"Broaden Your Views for Self-Supervised Video Learning","date":"2021-03-30","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-classification-with-finecoarse-networks","title":"Busy-Quiet Video Disentangling for Video Classification","date":"2021-03-29","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/videomoco-contrastive-video-representation","title":"VideoMoCo: Contrastive Video Representation Learning with Temporally Adversarial Examples","date":"2021-03-10","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tclr-temporal-contrastive-learning-for-video","title":"TCLR: Temporal Contrastive Learning for Video Representation","date":"2021-01-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/claster-clustering-with-reinforcement","title":"CLASTER: Clustering with Reinforcement Learning for Zero-Shot Action Recognition","date":"2021-01-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-relational-crosstransformers-for-few","title":"Temporal-Relational CrossTransformers for Few-Shot Action Recognition","date":"2021-01-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tensor-representations-for-action-recognition","title":"Tensor Representations for Action Recognition","date":"2020-12-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/smart-frame-selection-for-action-recognition","title":"SMART Frame Selection for Action Recognition","date":"2020-12-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bubblenet-a-disperse-recurrent-structure-to","title":"Bubblenet: A Disperse Recurrent Structure To Recognize Activities","date":"2020-10-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-representation-using","title":"Pretext-Contrastive Learning: Toward Good Practices in Self-supervised Video Representation Leaning","date":"2020-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rspnet-relative-speed-perception-for","title":"RSPNet: Relative Speed Perception for Unsupervised Video Representation Learning","date":"2020-10-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/depth-guided-adaptive-meta-fusion-network-for","title":"Depth Guided Adaptive Meta-Fusion Network for Few-shot Video Recognition","date":"2020-10-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-co-training-for-video","title":"Self-supervised Co-training for Video Representation Learning","date":"2020-10-19","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pose-and-joint-aware-action-recognition","title":"Pose And Joint-Aware Action Recognition","date":"2020-10-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/perf-net-pose-empowered-rgb-flow-net","title":"PERF-Net: Pose Empowered RGB-Flow Net","date":"2020-09-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spatiotemporal-contrastive-video","title":"Spatiotemporal Contrastive Video Representation Learning","date":"2020-08-09","rows_on_this_dataset":6,"code_links":4,"syntology":null},{"paper":"/paper/self-supervised-video-representation-learning-3","title":"Self-supervised Video Representation Learning Using Inter-intra Contrastive Framework","date":"2020-08-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/late-temporal-modeling-in-3d-cnn","title":"Late Temporal Modeling in 3D CNN Architectures with BERT for Action Recognition","date":"2020-08-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/motionsqueeze-neural-motion-feature-learning","title":"MotionSqueeze: Neural Motion Feature Learning for Video Understanding","date":"2020-07-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-multimodal-versatile-networks","title":"Self-Supervised MultiModal Versatile Networks","date":"2020-06-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/audio-visual-instance-discrimination-with","title":"Audio-Visual Instance Discrimination with Cross-Modal Agreement","date":"2020-04-27","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/omni-sourced-webly-supervised-learning-for","title":"Omni-sourced Webly-supervised Learning for Video Recognition","date":"2020-03-29","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":0,"samples_unverified":13,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporally-coherent-embeddings-for-self","title":"Temporally Coherent Embeddings for Self-Supervised Video Representation Learning","date":"2020-03-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-zero-shot-video-classification-end","title":"Rethinking Zero-shot Video Classification: End-to-end Training for Realistic Applications","date":"2020-03-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/evolving-losses-for-unsupervised-video","title":"Evolving Losses for Unsupervised Video Representation Learning","date":"2020-02-26","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/learning-spatio-temporal-representations-with","title":"Learning spatio-temporal representations with temporal squeeze pooling","date":"2020-02-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hallucinating-statistical-moment-and-subspace","title":"Self-supervising Action Recognition by Statistical Moment and Subspace Descriptors","date":"2020-01-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/few-shot-action-recognition-via-improved","title":"Few-shot Action Recognition with Permutation-invariant Attention","date":"2020-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-cloze-procedure-for-self-supervised","title":"Video Cloze Procedure for Self-Supervised Spatio-Temporal Learning","date":"2020-01-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-learning-by-cross-modal-audio","title":"Self-Supervised Learning by Cross-Modal Audio-Video Clustering","date":"2019-11-28","rows_on_this_dataset":5,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-representation-learning-by-dense","title":"Video Representation Learning by Dense Predictive Coding","date":"2019-09-10","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":1,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cooperative-cross-stream-network-for","title":"Cooperative Cross-Stream Network for Discriminative Action Representation","date":"2019-08-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/stm-spatiotemporal-and-motion-encoding-for","title":"STM: SpatioTemporal and Motion Encoding for Action Recognition","date":"2019-08-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/i-know-the-relationships-zero-shot-action","title":"I Know the Relationships: Zero-Shot Action Recognition via Two-Stream Graph Convolutional Networks and Knowledge Graphs","date":"2019-07-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/r-stan-residual-spatial-temporal-attention","title":"R-STAN: Residual Spatial-Temporal Attention Network for Action Recognition","date":"2019-06-19","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/learning-spatio-temporal-representation-with-3","title":"Learning Spatio-Temporal Representation with Local and Global Diffusion","date":"2019-06-13","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/hallucinating-bag-of-words-and-fisher-vector","title":"Hallucinating IDT Descriptors and I3D Optical Flow Features for Action Recognition with CNNs","date":"2019-06-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/faster-recurrent-networks-for-video","title":"FASTER Recurrent Networks for Efficient Video Classification","date":"2019-06-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-spatiotemporal-learning-via","title":"Self-Supervised Spatiotemporal Learning via Video Clip Order Prediction","date":"2019-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mars-motion-augmented-rgb-stream-for-action","title":"MARS: Motion-Augmented RGB Stream for Action Recognition","date":"2019-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-feature-aggregation-networks-for","title":"Hierarchical Feature Aggregation Networks for Video Action Recognition","date":"2019-05-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/holistic-large-scale-video-understanding","title":"Large Scale Holistic Video Understanding","date":"2019-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-spatio-temporal","title":"Self-supervised Spatio-temporal Representation Learning for Videos by Predicting Motion and Appearance Statistics","date":"2019-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/paying-more-attention-to-motion-attention","title":"Attention Distillation for Learning Video Representations","date":"2019-04-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/contextual-action-cues-from-camera-sensor-for","title":"Contextual Action Cues from Camera Sensor for Multi-Stream Action Recognition","date":"2019-03-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/distinit-learning-video-representations","title":"DistInit: Learning Video Representations Without a Single Labeled Video","date":"2019-01-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dmc-net-generating-discriminative-motion-cues","title":"DMC-Net: Generating Discriminative Motion Cues for Fast Compressed Video Action Recognition","date":"2019-01-11","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/d3d-distilled-3d-networks-for-video-action","title":"D3D: Distilled 3D Networks for Video Action Recognition","date":"2018-12-19","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/susinet-see-understand-and-summarize-it","title":"SUSiNet: See, Understand and Summarize it","date":"2018-12-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-spatiotemporal-feature","title":"Self-Supervised Spatiotemporal Feature Learning via Video Rotation Prediction","date":"2018-11-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-representation-learning","title":"Self-Supervised Video Representation Learning with Space-Time Cubic Puzzles","date":"2018-11-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/representation-flow-for-action-recognition","title":"Representation Flow for Action Recognition","date":"2018-10-02","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/learning-discriminative-video-representations","title":"Contrastive Video Representation Learning via Adversarial Perturbations","date":"2018-07-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cooperative-learning-of-audio-and-video","title":"Cooperative Learning of Audio and Video Models from Self-Supervised Synchronization","date":"2018-06-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-learning-of-motion-representation","title":"End-to-End Learning of Motion Representation for Video Understanding","date":"2018-04-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-universal-representation-for-unseen","title":"Towards Universal Representation for Unseen Action Recognition","date":"2018-03-22","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/rethinking-spatiotemporal-feature-learning","title":"Rethinking Spatiotemporal Feature Learning: Speed-Accuracy Trade-offs in Video Classification","date":"2017-12-13","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-closer-look-at-spatiotemporal-convolutions","title":"A Closer Look at Spatiotemporal Convolutions for Action Recognition","date":"2017-11-30","rows_on_this_dataset":6,"code_links":24,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/optical-flow-guided-feature-a-fast-and-robust","title":"Optical Flow Guided Feature: A Fast and Robust Motion Representation for Video Action Recognition","date":"2017-11-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/appearance-and-relation-networks-for-video","title":"Appearance-and-Relation Networks for Video Classification","date":"2017-11-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/convnet-architecture-search-for","title":"ConvNet Architecture Search for Spatiotemporal Feature Learning","date":"2017-08-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unsupervised-representation-learning-by","title":"Unsupervised Representation Learning by Sorting Sequences","date":"2017-08-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-action-recognition-with-error","title":"Zero-Shot Action Recognition With Error-Correcting Output Codes","date":"2017-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spatiotemporal-multiplier-networks-for-video","title":"Spatiotemporal Multiplier Networks for Video Action Recognition","date":"2017-07-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/alternative-semantic-representations-for-zero","title":"Alternative Semantic Representations for Zero-Shot Human Action Recognition","date":"2017-06-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/quo-vadis-action-recognition-a-new-model-and","title":"Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset","date":"2017-05-22","rows_on_this_dataset":6,"code_links":34,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":28,"samples_ran":16,"samples_unverified":12,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hidden-two-stream-convolutional-networks-for","title":"Hidden Two-Stream Convolutional Networks for Action Recognition","date":"2017-04-02","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/ts-lstm-and-temporal-inception-exploiting","title":"TS-LSTM and Temporal-Inception: Exploiting Spatiotemporal Dynamics for Activity Recognition","date":"2017-03-30","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/actionflownet-learning-motion-representation","title":"ActionFlowNet: Learning Motion Representation for Action Recognition","date":"2016-12-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multi-task-zero-shot-action-recognition-with","title":"Multi-Task Zero-Shot Action Recognition with Prioritised Data Augmentation","date":"2016-11-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spatiotemporal-residual-networks-for-video","title":"Spatiotemporal Residual Networks for Video Action Recognition","date":"2016-11-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/temporal-segment-networks-towards-good","title":"Temporal Segment Networks: Towards Good Practices for Deep Action Recognition","date":"2016-08-02","rows_on_this_dataset":1,"code_links":22,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":2,"samples_unverified":22,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dynamic-image-networks-for-action-recognition","title":"Dynamic Image Networks for Action Recognition","date":"2016-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/convolutional-two-stream-network-fusion-for","title":"Convolutional Two-Stream Network Fusion for Video Action Recognition","date":"2016-04-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/long-term-temporal-convolutions-for-action","title":"Long-term Temporal Convolutions for Action Recognition","date":"2016-04-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/shuffle-and-learn-unsupervised-learning-using","title":"Shuffle and Learn: Unsupervised Learning using Temporal Order Verification","date":"2016-03-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/objects2action-classifying-and-localizing","title":"Objects2action: Classifying and localizing actions without any video example","date":"2015-10-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/action-recognition-with-trajectory-pooled","title":"Action Recognition with Trajectory-Pooled Deep-Convolutional Descriptors","date":"2015-05-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-spatiotemporal-features-with-3d","title":"Learning Spatiotemporal Features with 3D Convolutional Networks","date":"2014-12-02","rows_on_this_dataset":1,"code_links":29,"syntology":null},{"paper":"/paper/evaluation-of-output-embeddings-for-fine","title":"Evaluation of Output Embeddings for Fine-Grained Image Classification","date":"2014-09-30","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/two-stream-convolutional-networks-for-action","title":"Two-Stream Convolutional Networks for Action Recognition in Videos","date":"2014-06-09","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":33,"samples_harvested":211,"samples_ran":70,"samples_unverified":141,"pointer_only_for_licence":65,"papers_with_no_sample_that_ran":8,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}