{"url":"/dataset/multithumos","name":"MultiTHUMOS","full_name":null,"description_markdown":"The **MultiTHUMOS** dataset contains dense, multilabel, frame-level action annotations for 30 hours across 400 videos in the THUMOS'14 action detection dataset. It consists of 38,690 annotations of 65 action classes, with an average of 1.5 labels per frame and 10.5 action classes per video.\r\n\r\nSource: [http://ai.stanford.edu/~syyeung/everymoment.html](http://ai.stanford.edu/~syyeung/everymoment.html)\r\nImage Source: [http://ai.stanford.edu/~syyeung/everymoment.html](http://ai.stanford.edu/~syyeung/everymoment.html)","description_withheld":null,"homepage":"http://ai.stanford.edu/~syyeung/everymoment.html","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/every-moment-counts-dense-detailed-labeling","title":"Every Moment Counts: Dense Detailed Labeling of Actions in Complex Videos","first_author":"Serena Yeung","url":null},"license":{"name":"CC BY 4.0","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Action Detection","url":"/task/action-detection","datasets_with_task":"/datasets/task/action-detection"}],"languages":[],"variants":["Multi-THUMOS","MultiTHUMOS"],"data_loaders":[],"num_papers_in_archive":58,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-detection-on-multi-thumos","task":"Action Detection","dataset_variant":"Multi-THUMOS","rows":8,"metrics":["mAP"],"first_row_in_archive_order":{"model":"MLAD","paper":"/paper/modeling-multi-label-action-dependencies-for","metrics":{"mAP":"51.5"},"code_links":[{"title":"ptirupat/MLAD","url":"https://github.com/ptirupat/MLAD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/temporal-action-localization-on-multithumos-1","task":"Temporal Action Localization","dataset_variant":"MultiTHUMOS","rows":8,"metrics":["Average mAP","mAP IOU@0.1","mAP IOU@0.2","mAP IOU@0.3","mAP IOU@0.4","mAP IOU@0.5","mAP IOU@0.6","mAP IOU@0.7","mAP IOU@0.8","mAP IOU@0.9"],"first_row_in_archive_order":{"model":"TriDet (VideoMAEv2)","paper":"/paper/temporal-action-localization-with-enhanced","metrics":{"Average mAP":"37.5","mAP IOU@0.2":"57.7","mAP IOU@0.5":"42.7","mAP IOU@0.7":"24.3"},"code_links":[{"title":"dingfengshi/tridet","url":"https://github.com/dingfengshi/tridet"},{"title":"sssste/tridet","url":"https://github.com/sssste/tridet"},{"title":"dingfengshi/tridetplus","url":"https://github.com/dingfengshi/tridetplus"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-detection-on-multithumos-1","task":"Action Detection","dataset_variant":"MultiTHUMOS","rows":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"PAT","paper":"/paper/pat-position-aware-transformer-for-dense","metrics":{"mAP":"44.6"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/dual-detrs-for-multi-label-temporal-action","title":"Dual DETRs for Multi-Label Temporal Action Detection","date":"2024-03-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-action-localization-with-enhanced","title":"Temporal Action Localization with Enhanced Instant Discriminability","date":"2023-09-11","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/pat-position-aware-transformer-for-dense","title":"PAT: Position-Aware Transformer for Dense Multi-Label Action Detection","date":"2023-08-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporalmaxer-maximize-temporal-context-with","title":"TemporalMaxer: Maximize Temporal Context with only Max Pooling for Temporal Action Localization","date":"2023-03-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pointtad-multi-label-temporal-action","title":"PointTAD: Multi-Label Temporal Action Detection with Learnable Query Points","date":"2022-10-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ms-tct-multi-scale-temporal-convtransformer","title":"MS-TCT: Multi-Scale Temporal ConvTransformer for Action Detection","date":"2021-12-07","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ctrn-class-temporal-relational-network-for","title":"CTRN: Class-Temporal Relational Network for Action Detection","date":"2021-10-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/modeling-multi-label-action-dependencies-for","title":"Modeling Multi-Label Action Dependencies for Temporal Action Localization","date":"2021-03-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pdan-pyramid-dilated-attention-network-for","title":"PDAN: Pyramid Dilated Attention Network for Action Detection","date":"2021-01-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/temporal-gaussian-mixture-layer-for-videos","title":"Temporal Gaussian Mixture Layer for Videos","date":"2018-03-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-latent-super-events-to-detect","title":"Learning Latent Super-Events to Detect Multiple Activities in Videos","date":"2017-12-05","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/every-moment-counts-dense-detailed-labeling","title":"Every Moment Counts: Dense Detailed Labeling of Actions in Complex Videos","date":"2015-07-21","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":14,"samples_ran":12,"samples_unverified":2,"pointer_only_for_licence":6,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}