{"url":"/dataset/ucf101-24","name":"UCF101-24","full_name":null,"description_markdown":"Click to add a brief description of the dataset (Markdown and LaTeX enabled).\r\n\r\nProvide:\r\n\r\n* a high-level explanation of the dataset characteristics\r\n* explain motivations and summary of its content\r\n* potential use cases of the dataset","description_withheld":null,"homepage":"","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Action Detection","url":"/task/action-detection","datasets_with_task":"/datasets/task/action-detection"},{"name":"Weakly-supervised Temporal Action Localization","url":"/task/weakly-supervised-temporal-action","datasets_with_task":"/datasets/task/weakly-supervised-temporal-action"},{"name":"Open Vocabulary Action Detection","url":"/task/open-vocabulary-action-detection","datasets_with_task":"/datasets/task/open-vocabulary-action-detection"},{"name":"Semi-Supervised Video Action Detection","url":"/task/semi-supervised-video-action-detection","datasets_with_task":"/datasets/task/semi-supervised-video-action-detection"}],"languages":[],"variants":[],"data_loaders":[],"num_papers_in_archive":17,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-detection-on-ucf101-24","task":"Action Detection","dataset_variant":"UCF101-24","rows":19,"metrics":["Frame-mAP 0.5","Video-mAP 0.1","Video-mAP 0.2","Video-mAP 0.5"],"first_row_in_archive_order":{"model":"STAR/L","paper":"/paper/end-to-end-spatio-temporal-action","metrics":{"Frame-mAP 0.5":"90.3","Video-mAP 0.2":"88.0","Video-mAP 0.5":"71.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-action-detection-on-ucf101-24","task":"Open Vocabulary Action Detection","dataset_variant":"UCF101-24","rows":1,"metrics":["val mAP"],"first_row_in_archive_order":{"model":"SiA","paper":"/paper/scaling-open-vocabulary-action-detection","metrics":{"val mAP":"42.6"},"code_links":[{"title":"siatheindochinese/sia_act_placeholder","url":"https://github.com/siatheindochinese/sia_act_placeholder"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-temporal-action-6","task":"Weakly-supervised Temporal Action Localization","dataset_variant":"UCF101-24","rows":1,"metrics":["mAP@0.2"],"first_row_in_archive_order":{"model":"Structured Keypoint Pooling","paper":"/paper/unified-keypoint-based-action-recognition","metrics":{"mAP@0.2":"61.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/scaling-open-vocabulary-action-detection","title":"Scaling Open-Vocabulary Action Detection","date":"2025-04-04","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/stable-mean-teacher-for-semi-supervised-video","title":"Stable Mean Teacher for Semi-supervised Video Action Detection","date":"2024-12-10","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-spatio-temporal-action","title":"End-to-End Spatio-Temporal Action Localisation with Video Transformers","date":"2023-04-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unified-keypoint-based-action-recognition","title":"Unified Keypoint-based Action Recognition Framework via Structured Keypoint Pooling","date":"2023-03-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/holistic-interaction-transformer-network-for","title":"Holistic Interaction Transformer Network for Action Detection","date":"2022-10-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-semi-supervised-learning-for-video","title":"End-to-End Semi-Supervised Learning for Video Action Detection","date":"2022-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/finding-action-tubes-with-a-sparse-to-dense","title":"Finding Action Tubes with a Sparse-to-Dense Framework","date":"2020-08-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/actions-as-moving-points","title":"Actions as Moving Points","date":"2020-01-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/you-only-watch-once-a-unified-cnn","title":"You Only Watch Once: A Unified CNN Architecture for Real-Time Spatiotemporal Action Localization","date":"2019-11-15","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":3,"samples_unverified":9,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hierarchical-self-attention-network-for","title":"Hierarchical Self-Attention Network for Action Localization in Videos","date":"2019-10-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/tacnet-transition-aware-context-network-for-1","title":"TACNet: Transition-Aware Context Network for Spatio-Temporal Action Detection","date":"2019-05-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/step-spatio-temporal-progressive-learning-for","title":"STEP: Spatio-Temporal Progressive Learning for Video Action Detection","date":"2019-04-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dance-with-flow-two-in-one-stream-action","title":"Dance with Flow: Two-in-One Stream Action Detection","date":"2019-04-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ava-a-video-dataset-of-spatio-temporally","title":"AVA: A Video Dataset of Spatio-temporally Localized Atomic Visual Actions","date":"2017-05-23","rows_on_this_dataset":1,"code_links":9,"syntology":null},{"paper":"/paper/tube-convolutional-neural-network-t-cnn-for","title":"Tube Convolutional Neural Network (T-CNN) for Action Detection in Videos","date":"2017-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-region-two-stream-r-cnn-for-action","title":"Multi-region two-stream R-CNN for action detection","date":"2016-09-17","rows_on_this_dataset":2,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":28,"samples_ran":9,"samples_unverified":19,"pointer_only_for_licence":2,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}