{"url":"/dataset/something-something-v1","name":"Something-Something V1","full_name":null,"description_markdown":"The 20BN-SOMETHING-SOMETHING dataset is a large collection of labeled video clips that show humans performing pre-defined basic actions with everyday objects. The dataset was created by a large number of crowd workers. It allows machine learning models to develop fine-grained understanding of basic actions that occur in the physical world. It contains 108,499 videos, with 86,017 in the training set, 11,522 in the validation set and 10,960 in the test set. There are 174 labels.\r\n\r\n⚠️ Attention: This is the outdated V1 of the dataset. V2 is available [here](https://paperswithcode.com/dataset/something-something-v2).\r\n\r\nSource: [https://20bn.com/datasets/something-something/v1](https://20bn.com/datasets/something-something/v1)\r\nImage Source: [https://20bn.com/datasets/something-something/v1](https://20bn.com/datasets/something-something/v1)","description_withheld":null,"homepage":"https://20bn.com/datasets/something-something/v1","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-something-something-video-database-for","title":"The \"something something\" video database for learning and evaluating visual common sense","first_author":"Raghav Goyal","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Action Recognition In Videos","url":"/task/action-recognition-in-videos-2","datasets_with_task":"/datasets/task/action-recognition-in-videos-2"},{"name":"Video Classification","url":"/task/video-classification","datasets_with_task":"/datasets/task/video-classification"}],"languages":[],"variants":["Something-Something V1"],"data_loaders":[{"repo":"https://github.com/open-mmlab/mmaction2","url":"https://github.com/open-mmlab/mmaction2/blob/master/tools/data/sthv1/README.md","frameworks":["pytorch"]}],"num_papers_in_archive":117,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-recognition-in-videos-on-something-1","task":"Action Recognition","dataset_variant":"Something-Something V1","rows":74,"metrics":["Top 1 Accuracy","Top 5 Accuracy","Param.","GFLOPs"],"first_row_in_archive_order":{"model":"InternVideo","paper":"/paper/internvideo-general-video-foundation-models","metrics":{"Top 1 Accuracy":"70.0"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"yingsen1/unimd","url":"https://github.com/yingsen1/unimd"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-something-2","task":"Action Recognition In Videos","dataset_variant":"Something-Something V1","rows":3,"metrics":["Top 1 Accuracy"],"first_row_in_archive_order":{"model":"STM (16 frames, ImageNet pretraining)","paper":"/paper/stm-spatiotemporal-and-motion-encoding-for","metrics":{"Top 1 Accuracy":"50.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-classification-on-something-something","task":"Video Classification","dataset_variant":"Something-Something V1","rows":1,"metrics":["Top-5 Accuracy"],"first_row_in_archive_order":{"model":"MSNet-R50En (ours)","paper":"/paper/motionsqueeze-neural-motion-feature-learning","metrics":{"Top-5 Accuracy":"84"},"code_links":[{"title":"arunos728/MotionSqueeze","url":"https://github.com/arunos728/MotionSqueeze"},{"title":"arunos728/arunos728.github.io","url":"https://github.com/arunos728/arunos728.github.io"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/tds-clip-temporal-difference-side-network-for","title":"TDS-CLIP: Temporal Difference Side Network for Image-to-Video Transfer Learning","date":"2024-08-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-correlation-structures-for-vision","title":"Learning Correlation Structures for Vision Transformers","date":"2024-04-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/side4video-spatial-temporal-side-network-for","title":"Side4Video: Spatial-Temporal Side Network for Memory-Efficient Image-to-Video Transfer Learning","date":"2023-11-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/temporally-adaptive-models-for-efficient","title":"Temporally-Adaptive Models for Efficient Video Understanding","date":"2023-08-10","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/what-can-simple-arithmetic-operations-do-for","title":"What Can Simple Arithmetic Operations Do for Temporal Modeling?","date":"2023-07-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videomae-v2-scaling-video-masked-autoencoders","title":"VideoMAE V2: Scaling Video Masked Autoencoders with Dual Masking","date":"2023-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-scale-motion-aware-module-for-video","title":"Multi-scale Motion-Aware Module for Video Action Recognition","date":"2023-02-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniformerv2-spatiotemporal-learning-by-arming","title":"UniFormerV2: Spatiotemporal Learning by Arming Image ViTs with Video UniFormer","date":"2022-09-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/spatial-temporal-pyramid-graph-reasoning-for","title":"Spatial-Temporal Pyramid Graph Reasoning for Action Recognition","date":"2022-08-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/spatiotemporal-self-attention-modeling-with","title":"Spatiotemporal Self-attention Modeling with Temporal Patch Shift for Action Recognition","date":"2022-07-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ae-net-adjoint-enhancement-network-for","title":"AE-Net:Adjoint Enhancement Network for Efficient Action Recognition in Video Understanding","date":"2022-07-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/action-recognition-with-motion","title":"Action Recognition With Motion Diversification and Dynamic Selection","date":"2022-07-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/stand-alone-inter-frame-attention-in-video-1","title":"Stand-Alone Inter-Frame Attention in Video Models","date":"2022-06-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mlp-3d-a-mlp-like-3d-architecture-with-1","title":"MLP-3D: A MLP-like 3D Architecture with Grouped Time Mixing","date":"2022-06-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/slow-fast-visual-tempo-learning-for-video","title":"Motion-driven Visual Tempo Learning for Video-based Action Recognition","date":"2022-02-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/action-keypoint-network-for-efficient-video","title":"Action Keypoint Network for Efficient Video Recognition","date":"2022-01-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/relational-self-attention-what-s-missing-in","title":"Relational Self-Attention: What's Missing in Attention for Video Understanding","date":"2021-11-02","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniformer-unified-transformer-for-efficient","title":"UniFormer: Unified Transformer for Efficient Spatial-Temporal Representation Learning","date":"2021-09-29","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/ean-event-adaptive-network-for-enhanced","title":"EAN: Event Adaptive Network for Enhanced Action Recognition","date":"2021-07-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ct-net-channel-tensorization-network-for-1","title":"CT-Net: Channel Tensorization Network for Video Classification","date":"2021-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":10,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-classification-with-finecoarse-networks","title":"Busy-Quiet Video Disentangling for Video Classification","date":"2021-03-29","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/learning-self-similarity-in-space-and-time-as-1","title":"Learning Self-Similarity in Space and Time as Generalized Motion for Video Action Recognition","date":"2021-02-14","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tdn-temporal-difference-networks-for","title":"TDN: Temporal Difference Networks for Efficient Action Recognition","date":"2020-12-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvfnet-multi-view-fusion-network-for","title":"MVFNet: Multi-View Fusion Network for Efficient Video Recognition","date":"2020-12-13","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/diverse-temporal-aggregation-and-depthwise","title":"Diverse Temporal Aggregation and Depthwise Spatiotemporal Factorization for Efficient Video Classification","date":"2020-12-01","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/pan-towards-fast-action-recognition-via","title":"PAN: Towards Fast Action Recognition via Learning Persistence of Appearance","date":"2020-08-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/motionsqueeze-neural-motion-feature-learning","title":"MotionSqueeze: Neural Motion Feature Learning for Video Understanding","date":"2020-07-20","rows_on_this_dataset":5,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/region-based-non-local-operation-for-video","title":"Region-based Non-local Operation for Video Classification","date":"2020-07-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/knowing-what-where-and-when-to-look-efficient","title":"Knowing What, Where and When to Look: Efficient Video Action Modeling with Attention","date":"2020-04-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gate-shift-networks-for-video-action","title":"Gate-Shift Networks for Video Action Recognition","date":"2019-12-01","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-reasoning-graph-for-activity","title":"Temporal Reasoning Graph for Activity Recognition","date":"2019-08-27","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/190807625","title":"Action recognition with spatial-temporal discriminative filter banks","date":"2019-08-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/stm-spatiotemporal-and-motion-encoding-for","title":"STM: SpatioTemporal and Motion Encoding for Action Recognition","date":"2019-08-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mars-motion-augmented-rgb-stream-for-action","title":"MARS: Motion-Augmented RGB Stream for Action Recognition","date":"2019-06-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-feature-aggregation-networks-for","title":"Hierarchical Feature Aggregation Networks for Video Action Recognition","date":"2019-05-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/recurrent-space-time-graphs-for-video","title":"Recurrent Space-time Graph Neural Networks","date":"2019-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-classification-with-channel-separated","title":"Video Classification with Channel-Separated Convolutional Networks","date":"2019-04-04","rows_on_this_dataset":5,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-shift-module-for-efficient-video","title":"TSM: Temporal Shift Module for Efficient Video Understanding","date":"2018-11-20","rows_on_this_dataset":3,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":6,"samples_unverified":10,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/motion-feature-network-fixed-motion-filter","title":"Motion Feature Network: Fixed Motion Filter for Action Recognition","date":"2018-07-26","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/videos-as-space-time-region-graphs","title":"Videos as Space-Time Region Graphs","date":"2018-06-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/eco-efficient-convolutional-network-for","title":"ECO: Efficient Convolutional Network for Online Video Understanding","date":"2018-04-24","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/moments-in-time-dataset-one-million-videos","title":"Moments in Time Dataset: one million videos for event understanding","date":"2018-01-09","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/rethinking-spatiotemporal-feature-learning","title":"Rethinking Spatiotemporal Feature Learning: Speed-Accuracy Trade-offs in Video Classification","date":"2017-12-13","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/temporal-relational-reasoning-in-videos","title":"Temporal Relational Reasoning in Videos","date":"2017-11-22","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/non-local-neural-networks","title":"Non-local Neural Networks","date":"2017-11-21","rows_on_this_dataset":1,"code_links":32,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":17,"samples_harvested":81,"samples_ran":45,"samples_unverified":36,"pointer_only_for_licence":19,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}