{"url":"/dataset/sports-1m","name":"Sports-1M","full_name":null,"description_markdown":"The **Sports-1M** dataset consists of over a million videos from YouTube. The videos in the dataset can be obtained through the YouTube URL specified by the authors. Approximately 7% (as of 2016) of the videos have been removed by the YouTube uploaders since the dataset was compiled. However, there are still over a million videos in the dataset with 487 sports-related categories with 1,000 to 3,000 videos per category. The videos are automatically labelled with 487 sports classes using the YouTube Topics API by analyzing the text metadata associated with the videos (e.g. tags, descriptions). Approximately 5% of the videos are annotated with more than one class.\r\n\r\nSource: [Review of Action Recognition and Detection Methods](https://arxiv.org/abs/1610.06906)\r\n\r\nImage Source: [Computer Vision for Sports](https://www.researchgate.net/publication/316477606_Computer_vision_for_sports_Current_applications_and_research_topics)","description_withheld":null,"homepage":"https://cs.stanford.edu/people/karpathy/deepvideo/","introduced_date":"2014-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/large-scale-video-classification-with-1","title":"Large-Scale Video Classification with Convolutional Neural Networks","first_author":"Andrej Karpathy","url":null},"license":{"name":"CC BY 3.0","url":"https://cs.stanford.edu/people/karpathy/deepvideo/#:~:is%20licensed%20under"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Action Recognition In Videos","url":"/task/action-recognition-in-videos-2","datasets_with_task":"/datasets/task/action-recognition-in-videos-2"}],"languages":[],"variants":["Sports-1M"],"data_loaders":[],"num_papers_in_archive":164,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-recognition-in-videos-on-sports-1m","task":"Action Recognition","dataset_variant":"Sports-1M","rows":9,"metrics":["Video hit@1 ","Video hit@5","Clip Hit@1"],"first_row_in_archive_order":{"model":"ip-CSN-152 (RGB)","paper":"/paper/video-classification-with-channel-separated","metrics":{"Video hit@1 ":"75.5","Video hit@5":"92.8"},"code_links":[{"title":"open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2"},{"title":"facebookresearch/R2Plus1D","url":"https://github.com/facebookresearch/R2Plus1D"},{"title":"facebookresearch/VMZ","url":"https://github.com/facebookresearch/VMZ"},{"title":"BB-Repos/BBaction","url":"https://github.com/BB-Repos/BBaction"},{"title":"salinasJJ/BBaction","url":"https://github.com/salinasJJ/BBaction"},{"title":"MindSpore-paper-code-2/code2","url":"https://github.com/MindSpore-paper-code-2/code2/tree/main/r2plus1d"},{"title":"Mind23-2/MindCode-62","url":"https://github.com/Mind23-2/MindCode-62"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-sports-1m-1","task":"Action Recognition In Videos","dataset_variant":"Sports-1M","rows":2,"metrics":["Video hit@1","Video hit@5"],"first_row_in_archive_order":{"model":"G-Blend","paper":"/paper/what-makes-training-multi-modal-networks-hard","metrics":{"Video hit@1":"74.8","Video hit@5":"92.4"},"code_links":[{"title":"facebookresearch/R2Plus1D","url":"https://github.com/facebookresearch/R2Plus1D"},{"title":"facebookresearch/VMZ","url":"https://github.com/facebookresearch/VMZ"},{"title":"guide2157/ChulaXrayClassifier","url":"https://github.com/guide2157/ChulaXrayClassifier"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/what-makes-training-multi-modal-networks-hard","title":"What Makes Training Multi-Modal Classification Networks Hard?","date":"2019-05-29","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":1,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-classification-with-channel-separated","title":"Video Classification with Channel-Separated Convolutional Networks","date":"2019-04-04","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-closer-look-at-spatiotemporal-convolutions","title":"A Closer Look at Spatiotemporal Convolutions for Action Recognition","date":"2017-11-30","rows_on_this_dataset":3,"code_links":24,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-spatio-temporal-representation-with","title":"Learning Spatio-Temporal Representation with Pseudo-3D Residual Networks","date":"2017-11-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/youtube-8m-a-large-scale-video-classification","title":"YouTube-8M: A Large-Scale Video Classification Benchmark","date":"2016-09-27","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/beyond-short-snippets-deep-networks-for-video","title":"Beyond Short Snippets: Deep Networks for Video Classification","date":"2015-03-31","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-spatiotemporal-features-with-3d","title":"Learning Spatiotemporal Features with 3D Convolutional Networks","date":"2014-12-02","rows_on_this_dataset":1,"code_links":29,"syntology":null},{"paper":"/paper/large-scale-video-classification-with-1","title":"Large-Scale Video Classification with Convolutional Neural Networks","date":"2014-06-23","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":32,"samples_ran":4,"samples_unverified":28,"pointer_only_for_licence":4,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}