{"url":"/dataset/hacs","name":"HACS","full_name":"Human Action Clips and Segments","description_markdown":"HACS is a dataset for human action recognition. It uses a taxonomy of 200 action classes, which is identical to that of the ActivityNet-v1.3 dataset. It has 504K videos retrieved from YouTube. Each one is strictly shorter than 4 minutes, and the average length is 2.6 minutes. A total of 1.5M clips of 2-second duration are sparsely sampled by methods based on both uniform randomness and consensus/disagreement of image classifiers. 0.6M and 0.9M clips are annotated as positive and negative samples, respectively.\r\n\r\nAuthors split the collection into training, validation and testing sets of size 1.4M, 50K and 50K clips, which are sampled\r\nfrom 492K, 6K and 6K videos, respectively.","description_withheld":null,"homepage":"http://hacs.csail.mit.edu/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/hacs-human-action-clips-and-segments-dataset","title":"HACS: Human Action Clips and Segments Dataset for Recognition and Temporal Localization","first_author":"Hang Zhao","url":null},"license":{"name":"BSD 3-Clause","url":"https://github.com/hangzhaomit/HACS-dataset/blob/master/LICENSE"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Action Localization","url":"/task/action-localization","datasets_with_task":"/datasets/task/action-localization"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"}],"languages":[],"variants":["HACS"],"data_loaders":[],"num_papers_in_archive":75,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/temporal-action-localization-on-hacs","task":"Temporal Action Localization","dataset_variant":"HACS","rows":12,"metrics":["Average-mAP","mAP@0.5","mAP@0.75","mAP@0.95"],"first_row_in_archive_order":{"model":"RDFA-S6 (InternVideo2-6B)","paper":"/paper/enhancing-temporal-action-localization","metrics":{"Average-mAP":"45.8","mAP@0.5":"66.4","mAP@0.75":"47.2","mAP@0.95":"14.3"},"code_links":[{"title":"lsy0882/RDFA-S6","url":"https://github.com/lsy0882/RDFA-S6"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-on-hacs","task":"Action Recognition","dataset_variant":"HACS","rows":8,"metrics":["Top 1 Accuracy","Top 5 Accuracy"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"Top 1 Accuracy":"97.0"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/enhancing-temporal-action-localization","title":"Enhancing Temporal Action Localization: Advanced S6 Modeling with Recurrent Mechanism","date":"2024-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dyfadet-dynamic-feature-aggregation-for","title":"DyFADet: Dynamic Feature Aggregation for Temporal Action Detection","date":"2024-07-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":6,"samples_unverified":6,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/video-mamba-suite-state-space-model-as-a","title":"Video Mamba Suite: State Space Model as a Versatile Alternative for Video Understanding","date":"2024-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-action-localization-with-enhanced","title":"Temporal Action Localization with Enhanced Instant Discriminability","date":"2023-09-11","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/tridet-temporal-action-detection-with","title":"TriDet: Temporal Action Detection with Relative Boundary Modeling","date":"2023-03-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniformerv2-spatiotemporal-learning-by-arming","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-09-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/low-fidelity-video-encoder-optimization-for","title":"Low-Fidelity Video Encoder Optimization for Temporal Action Localization","date":"2021-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-temporal-action-detection-with","title":"End-to-end Temporal Action Detection with Transformer","date":"2021-06-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learn-to-cycle-time-consistent-feature","title":"Learn to cycle: Time-consistent feature discovery for action recognition","date":"2020-06-15","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/hacs-human-action-clips-and-segments-dataset","title":"HACS: Human Action Clips and Segments Dataset for Recognition and Temporal Localization","date":"2017-12-26","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":47,"samples_ran":24,"samples_unverified":23,"pointer_only_for_licence":12,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}