{"url":"/dataset/charades","name":"Charades","full_name":null,"description_markdown":"The **Charades** dataset is composed of 9,848 videos of daily indoors activities with an average length of 30 seconds, involving interactions with 46 objects classes in 15 types of indoor scenes and containing a vocabulary of 30 verbs leading to 157 action classes. Each video in this dataset is annotated by multiple free-text descriptions, action labels, action intervals and classes of interacting objects. 267 different users were presented with a sentence, which includes objects and actions from a fixed vocabulary, and they recorded a video acting out the sentence. In total, the dataset contains 66,500 temporal annotations for 157 action classes, 41,104 labels for 46 object classes, and 27,847 textual descriptions of the videos. In the standard split there are7,986 training video and 1,863 validation video.\r\n\r\nSource: [Temporal Reasoning Graph for Activity Recognition](https://arxiv.org/abs/1908.09995)","description_withheld":null,"homepage":"http://vuchallenge.org/charades.html","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/hollywood-in-homes-crowdsourcing-data","title":"Hollywood in Homes: Crowdsourcing Data Collection for Activity Understanding","first_author":"Gunnar A. Sigurdsson","url":null},"license":{"name":"Custom (non-commercial)","url":"http://vuchallenge.org/license-charades.txt"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Action Detection","url":"/task/action-detection","datasets_with_task":"/datasets/task/action-detection"},{"name":"Action Classification","url":"/task/action-classification","datasets_with_task":"/datasets/task/action-classification"},{"name":"Video Understanding","url":"/task/video-understanding","datasets_with_task":"/datasets/task/video-understanding"},{"name":"Weakly Supervised Object Detection","url":"/task/weakly-supervised-object-detection","datasets_with_task":"/datasets/task/weakly-supervised-object-detection"},{"name":"Video Classification","url":"/task/video-classification","datasets_with_task":"/datasets/task/video-classification"},{"name":"Zero-Shot Action Recognition","url":"/task/zero-shot-action-recognition","datasets_with_task":"/datasets/task/zero-shot-action-recognition"}],"languages":[],"variants":["Charades"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/HuggingFaceM4/charades","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/aps/charades","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":428,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-classification-on-charades","task":"Action Classification","dataset_variant":"Charades","rows":49,"metrics":["MAP","FLOPs (G) x views"],"first_row_in_archive_order":{"model":"TokenLearner","paper":"/paper/tokenlearner-what-can-8-learned-tokens-do-for","metrics":{"MAP":"66.3"},"code_links":[{"title":"google-research/scenic","url":"https://github.com/google-research/scenic"},{"title":"keras-team/keras-io","url":"https://github.com/keras-team/keras-io/blob/master/examples/vision/token_learner.py"},{"title":"rish-16/tokenlearner-pytorch","url":"https://github.com/rish-16/tokenlearner-pytorch"},{"title":"ariG23498/TokenLearner","url":"https://github.com/ariG23498/TokenLearner"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/7/token_learner"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/4/token_learner"},{"title":"MindSpore-scientific/code-11","url":"https://github.com/MindSpore-scientific/code-11/tree/main/token_learner"},{"title":"MindSpore-scientific/code-1","url":"https://github.com/MindSpore-scientific/code-1/tree/main/token_learner"},{"title":"MindSpore-scientific/code-8","url":"https://github.com/MindSpore-scientific/code-8/tree/main/token_learner"},{"title":"MindSpore-scientific/code-10","url":"https://github.com/MindSpore-scientific/code-10/tree/main/token_learner"},{"title":"MindSpore-scientific-2/code-8","url":"https://github.com/MindSpore-scientific-2/code-8/tree/main/token_learner"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-detection-on-charades","task":"Action Detection","dataset_variant":"Charades","rows":16,"metrics":["mAP"],"first_row_in_archive_order":{"model":"TTM","paper":"/paper/token-turing-machines","metrics":{"mAP":"28.79"},"code_links":[{"title":"google-research/scenic","url":"https://github.com/google-research/scenic"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-object-detection-on-4","task":"Weakly Supervised Object Detection","dataset_variant":"Charades","rows":6,"metrics":["MAP"],"first_row_in_archive_order":{"model":"Spatial Prior","paper":"/paper/activity-driven-weakly-supervised-object","metrics":{"MAP":"10.03"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-action-recognition-on-charades-1","task":"Zero-Shot Action Recognition","dataset_variant":"Charades","rows":4,"metrics":["mAP"],"first_row_in_archive_order":{"model":"MSQNet","paper":"/paper/msqnet-actor-agnostic-action-recognition-with","metrics":{"mAP":"35.59"},"code_links":[{"title":"mondalanindya/msqnet","url":"https://github.com/mondalanindya/msqnet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-charades","task":"Action Recognition","dataset_variant":"Charades","rows":1,"metrics":["MAP"],"first_row_in_archive_order":{"model":"MSQNet","paper":"/paper/msqnet-actor-agnostic-action-recognition-with","metrics":{"MAP":"47.57"},"code_links":[{"title":"mondalanindya/msqnet","url":"https://github.com/mondalanindya/msqnet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-classification-on-charades","task":"Video Classification","dataset_variant":"Charades","rows":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"Multigrid","paper":"/paper/a-multigrid-method-for-efficiently-training","metrics":{"mAP":"38.2"},"code_links":[{"title":"facebookresearch/SlowFast","url":"https://github.com/facebookresearch/SlowFast"},{"title":"kkahatapitiya/X3D-Multigrid","url":"https://github.com/kkahatapitiya/X3D-Multigrid"},{"title":"alexandrosstergiou/Squeeze-and-Recursion-Temporal-Gates","url":"https://github.com/alexandrosstergiou/Squeeze-and-Recursion-Temporal-Gates"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/unimd-towards-unifying-moment-retrieval-and","title":"UniMD: Towards Unifying Moment Retrieval and Temporal Action Detection","date":"2024-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adafocus-towards-end-to-end-weakly-supervised","title":"Towards Weakly Supervised End-to-end Learning for Long-video Action Recognition","date":"2023-11-28","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/pat-position-aware-transformer-for-dense","title":"PAT: Position-Aware Transformer for Dense Multi-Label Action Detection","date":"2023-08-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/msqnet-actor-agnostic-action-recognition-with","title":"Actor-agnostic Multi-label Action Recognition with Multi-modal Query","date":"2023-07-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/victr-video-conditioned-text-representations","title":"VicTR: Video-conditioned Text Representations for Activity Recognition","date":"2023-04-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/match-expand-and-improve-unsupervised","title":"MAtch, eXpand and Improve: Unsupervised Finetuning for Zero-Shot Action Recognition with Language Knowledge","date":"2023-03-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bidirectional-cross-modal-knowledge","title":"Bidirectional Cross-Modal Knowledge Exploration for Video Recognition with Pre-trained Vision-Language Models","date":"2022-12-31","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/rethinking-video-vits-sparse-video-tubes-for","title":"Rethinking Video ViTs: Sparse Video Tubes for Joint Image and Video Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/token-turing-machines","title":"Token Turing Machines","date":"2022-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-clip-hitchhiker-s-guide-to-long-video","title":"A CLIP-Hitchhiker's Guide to Long Video Retrieval","date":"2022-05-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ms-tct-multi-scale-temporal-convtransformer","title":"MS-TCT: Multi-Scale Temporal ConvTransformer for Action Detection","date":"2021-12-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-pretraining-with","title":"Weakly-guided Self-supervised Pretraining for Temporal Activity Detection","date":"2021-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-spatio-temporal-layouts-for","title":"Revisiting spatio-temporal layouts for compositional action recognition","date":"2021-11-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ctrn-class-temporal-relational-network-for","title":"CTRN: Class-Temporal Relational Network for Action Detection","date":"2021-10-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/actionclip-a-new-paradigm-for-video-action","title":"ActionCLIP: A New Paradigm for Video Action Recognition","date":"2021-09-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":5,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tokenlearner-what-can-8-learned-tokens-do-for","title":"TokenLearner: What Can 8 Learned Tokens Do for Images and Videos?","date":"2021-06-21","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/continual-3d-convolutional-neural-networks","title":"Continual 3D Convolutional Neural Networks for Real-time Processing of Videos","date":"2021-05-31","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/vidtr-video-transformer-without-convolutions","title":"VidTr: Video Transformer Without Convolutions","date":"2021-04-23","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/multiscale-vision-transformers","title":"Multiscale Vision Transformers","date":"2021-04-22","rows_on_this_dataset":6,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":13,"samples_unverified":13,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/movinets-mobile-video-networks-for-efficient","title":"MoViNets: Mobile Video Networks for Efficient Video Recognition","date":"2021-03-21","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/modeling-multi-label-action-dependencies-for","title":"Modeling Multi-Label Action Dependencies for Temporal Action Localization","date":"2021-03-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coarse-fine-networks-for-temporal-activity","title":"Coarse-Fine Networks for Temporal Activity Detection in Videos","date":"2021-03-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pdan-pyramid-dilated-attention-network-for","title":"PDAN: Pyramid Dilated Attention Network for Action Detection","date":"2021-01-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pose-and-joint-aware-action-recognition","title":"Pose And Joint-Aware Action Recognition","date":"2020-10-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/assemblenet-assembling-modality","title":"AssembleNet++: Assembling Modality Representations via Attention Connections","date":"2020-08-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/avid-dataset-anonymized-videos-from-diverse","title":"AViD Dataset: Anonymized Videos from Diverse Countries","date":"2020-07-10","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hallucinating-statistical-moment-and-subspace","title":"Self-supervising Action Recognition by Statistical Moment and Subspace Descriptors","date":"2020-01-14","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/a-multigrid-method-for-efficiently-training","title":"A Multigrid Method for Efficiently Training Video Models","date":"2019-12-02","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hallucinating-bag-of-words-and-fisher-vector","title":"Hallucinating IDT Descriptors and I3D Optical Flow Features for Action Recognition with CNNs","date":"2019-06-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pa3d-pose-action-3d-machine-for-video","title":"PA3D: Pose-Action 3D Machine for Video Recognition","date":"2019-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/assemblenet-searching-for-multi-stream-neural","title":"AssembleNet: Searching for Multi-Stream Neural Connectivity in Video Architectures","date":"2019-05-30","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/neural-message-passing-on-hybrid-spatio","title":"Representation Learning on Visual-Symbolic Graphs for Video Understanding","date":"2019-05-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/activity-driven-weakly-supervised-object","title":"Activity Driven Weakly Supervised Object Detection","date":"2019-04-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/long-term-feature-banks-for-detailed-video","title":"Long-Term Feature Banks for Detailed Video Understanding","date":"2018-12-12","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/slowfast-networks-for-video-recognition","title":"SlowFast Networks for Video Recognition","date":"2018-12-10","rows_on_this_dataset":3,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/timeception-for-complex-action-recognition","title":"Timeception for Complex Action Recognition","date":"2018-12-04","rows_on_this_dataset":3,"code_links":3,"syntology":null},{"paper":"/paper/evolving-space-time-neural-architectures-for","title":"Evolving Space-Time Neural Architectures for Videos","date":"2018-11-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pcl-proposal-cluster-learning-for-weakly","title":"PCL: Proposal Cluster Learning for Weakly Supervised Object Detection","date":"2018-07-09","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videos-as-space-time-region-graphs","title":"Videos as Space-Time Region Graphs","date":"2018-06-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/potion-pose-motion-representation-for-action","title":"PoTion: Pose MoTion Representation for Action Recognition","date":"2018-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-gaussian-mixture-layer-for-videos","title":"Temporal Gaussian Mixture Layer for Videos","date":"2018-03-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-latent-super-events-to-detect","title":"Learning Latent Super-Events to Detect Multiple Activities in Videos","date":"2017-12-05","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/compressed-video-action-recognition","title":"Compressed Video Action Recognition","date":"2017-12-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/temporal-relational-reasoning-in-videos","title":"Temporal Relational Reasoning in Videos","date":"2017-11-22","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-dynamic-graph-lstm-for-action-driven","title":"Temporal Dynamic Graph LSTM for Action-driven Video Object Detection","date":"2017-08-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/quo-vadis-action-recognition-a-new-model-and","title":"Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset","date":"2017-05-22","rows_on_this_dataset":1,"code_links":34,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":28,"samples_ran":16,"samples_unverified":12,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/r-c3d-region-convolutional-3d-network-for","title":"R-C3D: Region Convolutional 3D Network for Temporal Activity Detection","date":"2017-03-22","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/asynchronous-temporal-fields-for-action","title":"Asynchronous Temporal Fields for Action Recognition","date":"2016-12-19","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/contextlocnet-context-aware-deep-network","title":"ContextLocNet: Context-Aware Deep Network Models for Weakly Supervised Localization","date":"2016-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-deep-detection-networks","title":"Weakly Supervised Deep Detection Networks","date":"2015-11-09","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contextual-action-recognition-with-rcnn","title":"Contextual Action Recognition with R*CNN","date":"2015-05-05","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/two-stream-convolutional-networks-for-action","title":"Two-Stream Convolutional Networks for Action Recognition in Videos","date":"2014-06-09","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":18,"samples_harvested":152,"samples_ran":64,"samples_unverified":88,"pointer_only_for_licence":31,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}