{"url":"/dataset/kth","name":"KTH","full_name":"KTH Action dataset","description_markdown":"The efforts to create a non-trivial and publicly available dataset for action recognition was initiated at the **KTH** Royal Institute of Technology in 2004. The KTH dataset is one of the most standard datasets, which contains six actions: walk, jog, run, box, hand-wave, and hand clap. To account for performance nuance, each action is performed by 25 different individuals, and the setting is systematically altered for each action per actor. Setting variations include: outdoor (s1), outdoor with scale variation (s2), outdoor with different clothes (s3), and indoor (s4). These variations test the ability of each algorithm to identify actions independent of the background, appearance of the actors, and the scale of the actors.\r\n\r\nSource: [Review of Action Recognition and Detection Methods](https://arxiv.org/abs/1610.06906)","description_withheld":null,"homepage":"https://www.csc.kth.se/cvap/actions/","introduced_date":"2004-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"Recognizing Human Actions: A Local SVM Approach","first_author":null,"url":"https://doi.org/10.1109/ICPR.2004.1334462"},"license":{"name":"Custom (non-commercial, attribution)","url":"https://www.csc.kth.se/cvap/actions/#:~:text=publicly%20available"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Video Prediction","url":"/task/video-prediction","datasets_with_task":"/datasets/task/video-prediction"}],"languages":[],"variants":["KTH"],"data_loaders":[{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/kth-actions-dataset","frameworks":["tf","pytorch"]}],"num_papers_in_archive":279,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-prediction-on-kth","task":"Video Prediction","dataset_variant":"KTH","rows":31,"metrics":["FVD","SSIM","PSNR","LPIPS","Cond","Train","Pred","Params (M)","MSE","Diversity"],"first_row_in_archive_order":{"model":"Grid-keypoints","paper":"/paper/accurate-grid-keypoint-learning-for-efficient","metrics":{"Cond":"10","FVD":"144.2","LPIPS":"0.092","PSNR":"27.11","Params (M)":"2.0","Pred":"40","SSIM":"0.837","Train":"10"},"code_links":[{"title":"xjgaocs/Grid-Keypoint-Learning","url":"https://github.com/xjgaocs/Grid-Keypoint-Learning"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-on-kth","task":"Action Recognition","dataset_variant":"KTH","rows":1,"metrics":["16:9 Accuracy"],"first_row_in_archive_order":{"model":"CNN-GRU","paper":"/paper/temporal-relations-of-informative-frames-in","metrics":{"16:9 Accuracy":"95.38"},"code_links":[{"title":"Alirezarahnamaa/Temporal-Relations-of-Informative-Frames-in-Action-Recognition","url":"https://github.com/Alirezarahnamaa/Temporal-Relations-of-Informative-Frames-in-Action-Recognition"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/temporal-relations-of-informative-frames-in","title":"Temporal Relations of Informative Frames in Action Recognition","date":"2024-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-prediction-at-multiple-scales-with","title":"MSPred: Video Prediction at Multiple Spatio-Temporal Scales with Hierarchical Recurrent Networks","date":"2022-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/slamp-stochastic-latent-appearance-and-motion","title":"SLAMP: Stochastic Latent Appearance and Motion Prediction","date":"2021-08-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/accurate-grid-keypoint-learning-for-efficient","title":"Accurate Grid Keypoint Learning for Efficient Video Prediction","date":"2021-07-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diverse-video-generation-using-a-gaussian-1","title":"Diverse Video Generation using a Gaussian Process Trigger","date":"2021-07-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-prediction-recalling-long-term-motion","title":"Video Prediction Recalling Long-term Motion Context via Memory Alignment Learning","date":"2021-04-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/predrnn-a-recurrent-neural-network-for","title":"PredRNN: A Recurrent Neural Network for Spatiotemporal Predictive Learning","date":"2021-03-17","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/exploring-spatial-temporal-multi-frequency","title":"Exploring Spatial-Temporal Multi-Frequency Analysis for High-Fidelity and Temporal-Consistency Video Prediction","date":"2020-02-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stochastic-latent-residual-video-prediction-1","title":"Stochastic Latent Residual Video Prediction","date":"2020-02-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/convolutional-tensor-train-lstm-for-spatio","title":"Convolutional Tensor-Train LSTM for Spatio-temporal Learning","date":"2020-02-21","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/z-order-recurrent-neural-networks-for-video","title":"Z-Order Recurrent Neural Networks for Video Prediction","date":"2019-07-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unsupervised-learning-of-object-structure-and","title":"Unsupervised Learning of Object Structure and Dynamics from Videos","date":"2019-06-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/eidetic-3d-lstm-a-model-for-video-prediction","title":"Eidetic 3D LSTM: A Model for Video Prediction and Beyond","date":"2019-05-01","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/varnet-exploring-variations-for-unsupervised","title":"VarNet: Exploring Variations for Unsupervised Video Prediction","date":"2018-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/predrnn-towards-a-resolution-of-the-deep-in","title":"PredRNN++: Towards A Resolution of the Deep-in-Time Dilemma in Spatiotemporal Predictive Learning","date":"2018-04-17","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/msnet-mutual-suppression-network-for","title":"Mutual Suppression Network for Video Prediction using Disentangled Features","date":"2018-04-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/stochastic-adversarial-video-prediction","title":"Stochastic Adversarial Video Prediction","date":"2018-04-04","rows_on_this_dataset":4,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":4,"samples_unverified":12,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stochastic-video-generation-with-a-learned","title":"Stochastic Video Generation with a Learned Prior","date":"2018-02-21","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/folded-recurrent-neural-networks-for-future","title":"Folded Recurrent Neural Networks for Future Video Prediction","date":"2017-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stochastic-variational-video-prediction","title":"Stochastic Variational Video Prediction","date":"2017-10-30","rows_on_this_dataset":3,"code_links":3,"syntology":null},{"paper":"/paper/decomposing-motion-and-content-for-natural","title":"Decomposing Motion and Content for Natural Video Sequence Prediction","date":"2017-06-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/deep-learning-for-precipitation-nowcasting-a","title":"Deep Learning for Precipitation Nowcasting: A Benchmark and A New Model","date":"2017-06-12","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":9,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-pixel-networks","title":"Video Pixel Networks","date":"2016-10-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dynamic-filter-networks","title":"Dynamic Filter Networks","date":"2016-05-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/convolutional-lstm-network-a-machine-learning","title":"Convolutional LSTM Network: A Machine Learning Approach for Precipitation Nowcasting","date":"2015-06-13","rows_on_this_dataset":1,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":57,"samples_ran":23,"samples_unverified":34,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}