{"url":"/dataset/kinetics-700","name":"Kinetics-700","full_name":"Kinetics-700","description_markdown":"Kinetics-700 is a video dataset of 650,000 clips that covers 700 human action classes. The videos include human-object interactions such as playing instruments, as well as human-human interactions such as shaking hands and hugging. Each action class has at least 700 video clips. Each clip is annotated with an action class and lasts approximately 10 seconds.","description_withheld":null,"homepage":"https://deepmind.com/research/open-source/kinetics","introduced_date":"2019-07-15","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-short-note-on-the-kinetics-700-human-action","title":"A Short Note on the Kinetics-700 Human Action Dataset","first_author":"Joao Carreira","url":null},"license":{"name":"Commons Attribution 4.0 International License","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Image Clustering","url":"/task/image-clustering","datasets_with_task":"/datasets/task/image-clustering"},{"name":"Action Classification","url":"/task/action-classification","datasets_with_task":"/datasets/task/action-classification"},{"name":"Video Generation","url":"/task/video-generation","datasets_with_task":"/datasets/task/video-generation"},{"name":"Semantic Object Interaction Classification","url":"/task/semantic-object-interaction-classification","datasets_with_task":"/datasets/task/semantic-object-interaction-classification"}],"languages":[],"variants":["Kinetics-700"],"data_loaders":[{"repo":"https://github.com/voxel51/fiftyone","url":"https://docs.voxel51.com/user_guide/dataset_zoo/datasets.html#kinetics-700","frameworks":["tf","pytorch"]},{"repo":"https://github.com/open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2/blob/master/tools/data/kinetics/README.md","frameworks":["pytorch"]}],"num_papers_in_archive":95,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-classification-on-kinetics-700","task":"Action Classification","dataset_variant":"Kinetics-700","rows":36,"metrics":["Top-1 Accuracy","Top-5 Accuracy"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"Top-1 Accuracy":"85.9"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-clustering-on-kinetics-700","task":"Image Clustering","dataset_variant":"Kinetics-700","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"TURTLE (CLIP + DINOv2)","paper":"/paper/let-go-of-your-labels-with-unsupervised-1","metrics":{"Accuracy":"43.0"},"code_links":[{"title":"mlbio-epfl/turtle","url":"https://github.com/mlbio-epfl/turtle"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-generation-on-kinetics-700","task":"Video Generation","dataset_variant":"Kinetics-700","rows":1,"metrics":["FID","FVD"],"first_row_in_archive_order":{"model":"DiT-XL/2 + CVAE-FT-SE","paper":"/paper/improving-the-diffusability-of-autoencoders","metrics":{"FID":"8.59","FVD":"135.15"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/improving-the-diffusability-of-autoencoders","title":"Improving the Diffusability of Autoencoders","date":"2025-02-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/let-go-of-your-labels-with-unsupervised-1","title":"Let Go of Your Labels with Unsupervised Transfer","date":"2024-06-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/hiera-a-hierarchical-vision-transformer","title":"Hiera: A Hierarchical Vision Transformer without the Bells-and-Whistles","date":"2023-06-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/aim-adapting-image-models-for-efficient-video","title":"AIM: Adapting Image Models for Efficient Video Action Recognition","date":"2023-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-video-vits-sparse-video-tubes-for","title":"Rethinking Video ViTs: Sparse Video Tubes for Joint Image and Video Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniformerv2-spatiotemporal-learning-by-arming","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-09-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-models-are-more-robust-and-fair-when","title":"Vision Models Are More Robust And Fair When Pretrained On Uncurated Images Without Supervision","date":"2022-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multiview-transformers-for-video-recognition","title":"Multiview Transformers for Video Recognition","date":"2022-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/masked-feature-prediction-for-self-supervised","title":"Masked Feature Prediction for Self-Supervised Visual Pre-Training","date":"2021-12-16","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/co-training-transformer-with-videos-and","title":"Co-training Transformer with Videos and Images Improves Action Recognition","date":"2021-12-14","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/improved-multiscale-vision-transformers-for","title":"MViTv2: Improved Multiscale Vision Transformers for Classification and Detection","date":"2021-12-02","rows_on_this_dataset":3,"code_links":9,"syntology":null},{"paper":"/paper/vidtr-video-transformer-without-convolutions","title":"VidTr: Video Transformer Without Convolutions","date":"2021-04-23","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/movinets-mobile-video-networks-for-efficient","title":"MoViNets: Mobile Video Networks for Efficient Video Recognition","date":"2021-03-21","rows_on_this_dataset":7,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learn-to-cycle-time-consistent-feature","title":"Learn to cycle: Time-consistent feature discovery for action recognition","date":"2020-06-15","rows_on_this_dataset":5,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":76,"samples_ran":37,"samples_unverified":39,"pointer_only_for_licence":4,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}