{"url":"/dataset/kinetics-600","name":"Kinetics-600","full_name":null,"description_markdown":"The **Kinetics-600** is a large-scale action recognition dataset which consists of around 480K videos from 600 action categories. The 480K videos are divided into 390K, 30K, 60K for training, validation and test sets, respectively. Each video in the dataset is a 10-second clip of action moment annotated from raw YouTube video. It is an extensions of the Kinetics-400 dataset.\r\n\r\nSource: [Learning to Localize Actions from Moments](https://arxiv.org/abs/2008.13705)\r\nImage Source: [https://towardsdatascience.com/downloading-the-kinetics-dataset-for-human-action-recognition-in-deep-learning-500c3d50f776](https://towardsdatascience.com/downloading-the-kinetics-dataset-for-human-action-recognition-in-deep-learning-500c3d50f776)","description_withheld":null,"homepage":"https://deepmind.com/research/open-source/kinetics","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-short-note-about-kinetics-600","title":"A Short Note about Kinetics-600","first_author":"Joao Carreira","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Classification","url":"/task/action-classification","datasets_with_task":"/datasets/task/action-classification"},{"name":"Action Recognition In Videos","url":"/task/action-recognition-in-videos-2","datasets_with_task":"/datasets/task/action-recognition-in-videos-2"},{"name":"Video Prediction","url":"/task/video-prediction","datasets_with_task":"/datasets/task/video-prediction"},{"name":"Video Generation","url":"/task/video-generation","datasets_with_task":"/datasets/task/video-generation"},{"name":"Self-Supervised Action Recognition","url":"/task/self-supervised-action-recognition","datasets_with_task":"/datasets/task/self-supervised-action-recognition"}],"languages":[],"variants":["Kinetics-600 12 frames, 64x64","Kinetics-600 48 frames, 64x64","Kinetics-600 12 frames, 128x128","Kinetics-600"],"data_loaders":[{"repo":"https://github.com/voxel51/fiftyone","url":"https://docs.voxel51.com/user_guide/dataset_zoo/datasets.html#kinetics-600","frameworks":["tf","pytorch"]},{"repo":"https://github.com/open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2/blob/master/tools/data/kinetics/README.md","frameworks":["pytorch"]}],"num_papers_in_archive":148,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-classification-on-kinetics-600","task":"Action Classification","dataset_variant":"Kinetics-600","rows":65,"metrics":["Top-1 Accuracy","Top-5 Accuracy","GFLOPs"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"Top-1 Accuracy":"91.9"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/self-supervised-action-recognition-on","task":"Self-Supervised Action Recognition","dataset_variant":"Kinetics-600","rows":5,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"CVRL (R3D-152 2x)","paper":"/paper/spatiotemporal-contrastive-video","metrics":{"Top-1 Accuracy":"72.9"},"code_links":[{"title":"tensorflow/models","url":"https://github.com/tensorflow/models/tree/master/official/projects/video_ssl"},{"title":"ed-fish/spatio-temporal-contrastive-film","url":"https://github.com/ed-fish/spatio-temporal-contrastive-film"},{"title":"ed-fish/spatio-temporal-contrastive-video","url":"https://github.com/ed-fish/spatio-temporal-contrastive-video"},{"title":"applecrumble123/CVLR_pytorch","url":"https://github.com/applecrumble123/CVLR_pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-kinetics-600","task":"Action Recognition In Videos","dataset_variant":"Kinetics-600","rows":1,"metrics":["Top-1 Accuracy","Top-5 Accuracy"],"first_row_in_archive_order":{"model":"Florence","paper":"/paper/florence-a-new-foundation-model-for-computer","metrics":{"Top-1 Accuracy":"87.8","Top-5 Accuracy":"97.8"},"code_links":[{"title":"microsoft/unicl","url":"https://github.com/microsoft/unicl"},{"title":"MindCode-4/code-3","url":"https://github.com/MindCode-4/code-3/tree/main/florence2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/hiera-a-hierarchical-vision-transformer","title":"Hiera: A Hierarchical Vision Transformer without the Bells-and-Whistles","date":"2023-06-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videomae-v2-scaling-video-masked-autoencoders","title":"VideoMAE V2: Scaling Video Masked Autoencoders with Dual Masking","date":"2023-03-29","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-video-vits-sparse-video-tubes-for","title":"Rethinking Video ViTs: Sparse Video Tubes for Joint Image and Video Learning","date":"2022-12-06","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniformerv2-spatiotemporal-learning-by-arming","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-09-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/expanding-language-image-pretrained-models","title":"Expanding Language-Image Pretrained Models for General Video Recognition","date":"2022-08-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multiview-transformers-for-video-recognition","title":"Multiview Transformers for Video Recognition","date":"2022-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/merlot-reserve-neural-script-knowledge","title":"MERLOT Reserve: Neural Script Knowledge through Vision and Language and Sound","date":"2022-01-07","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/masked-feature-prediction-for-self-supervised","title":"Masked Feature Prediction for Self-Supervised Visual Pre-Training","date":"2021-12-16","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/co-training-transformer-with-videos-and","title":"Co-training Transformer with Videos and Images Improves Action Recognition","date":"2021-12-14","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/improved-multiscale-vision-transformers-for","title":"MViTv2: Improved Multiscale Vision Transformers for Classification and Detection","date":"2021-12-02","rows_on_this_dataset":4,"code_links":9,"syntology":null},{"paper":"/paper/florence-a-new-foundation-model-for-computer","title":"Florence: A New Foundation Model for Computer Vision","date":"2021-11-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/uniformer-unified-transformer-for-efficient","title":"UniFormer: Unified Transformer for Efficient Spatial-Temporal Representation Learning","date":"2021-09-29","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/revisiting-3d-resnets-for-video-recognition","title":"Revisiting 3D ResNets for Video Recognition","date":"2021-09-03","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/video-swin-transformer","title":"Video Swin Transformer","date":"2021-06-24","rows_on_this_dataset":2,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":32,"samples_ran":7,"samples_unverified":25,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tokenlearner-what-can-8-learned-tokens-do-for","title":"TokenLearner: What Can 8 Learned Tokens Do for Images and Videos?","date":"2021-06-21","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/space-time-mixing-attention-for-video","title":"Space-time Mixing Attention for Video Transformer","date":"2021-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vatt-transformers-for-multimodal-self","title":"VATT: Transformers for Multimodal Self-Supervised Learning from Raw Video, Audio and Text","date":"2021-04-22","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multiscale-vision-transformers","title":"Multiscale Vision Transformers","date":"2021-04-22","rows_on_this_dataset":3,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":13,"samples_unverified":13,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/broaden-your-views-for-self-supervised-video","title":"Broaden Your Views for Self-Supervised Video Learning","date":"2021-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/2103-15691","title":"ViViT: A Video Vision Transformer","date":"2021-03-29","rows_on_this_dataset":3,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":14,"samples_unverified":7,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/movinets-mobile-video-networks-for-efficient","title":"MoViNets: Mobile Video Networks for Efficient Video Recognition","date":"2021-03-21","rows_on_this_dataset":8,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/perf-net-pose-empowered-rgb-flow-net","title":"PERF-Net: Pose Empowered RGB-Flow Net","date":"2020-09-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spatiotemporal-contrastive-video","title":"Spatiotemporal Contrastive Video Representation Learning","date":"2020-08-09","rows_on_this_dataset":3,"code_links":4,"syntology":null},{"paper":"/paper/self-supervised-multimodal-versatile-networks","title":"Self-Supervised MultiModal Versatile Networks","date":"2020-06-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-spatio-temporal-representation-with-3","title":"Learning Spatio-Temporal Representation with Local and Global Diffusion","date":"2019-06-13","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/d3d-distilled-3d-networks-for-video-action","title":"D3D: Distilled 3D Networks for Video Action Recognition","date":"2018-12-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/slowfast-networks-for-video-recognition","title":"SlowFast Networks for Video Recognition","date":"2018-12-10","rows_on_this_dataset":5,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-short-note-about-kinetics-600","title":"A Short Note about Kinetics-600","date":"2018-08-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-spatiotemporal-feature-learning","title":"Rethinking Spatiotemporal Feature Learning: Speed-Accuracy Trade-offs in Video Classification","date":"2017-12-13","rows_on_this_dataset":3,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":18,"samples_harvested":190,"samples_ran":82,"samples_unverified":108,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}