{"url":"/dataset/epic-kitchens","name":"EPIC-KITCHENS-55","full_name":null,"description_markdown":"The EPIC-KITCHENS-55 dataset comprises a set of 432 egocentric videos recorded by 32 participants in their kitchens at 60fps with a head mounted camera. There is no guiding script for the participants who freely perform activities in kitchens related to cooking, food preparation or washing up among others. Each video is split into short action segments (mean duration is 3.7s) with specific start and end times and a verb and noun annotation describing the action (e.g. ‘open fridge‘). The verb classes are 125 and the noun classes 331. The dataset is divided into one train and two test splits.\r\n\r\nSource: [Egocentric Hand Track and Object-based Human Action Recognition](https://arxiv.org/abs/1905.00742)\r\nImage Source: [https://epic-kitchens.github.io/2020-100](https://epic-kitchens.github.io/2020-100)","description_withheld":null,"homepage":"https://epic-kitchens.github.io/2019","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/scaling-egocentric-vision-the-epic-kitchens","title":"Scaling Egocentric Vision: The EPIC-KITCHENS Dataset","first_author":"Dima Damen","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Video Understanding","url":"/task/video-understanding","datasets_with_task":"/datasets/task/video-understanding"},{"name":"Video Object Detection","url":"/task/video-object-detection","datasets_with_task":"/datasets/task/video-object-detection"},{"name":"Egocentric Activity Recognition","url":"/task/egocentric-activity-recognition","datasets_with_task":"/datasets/task/egocentric-activity-recognition"}],"languages":[],"variants":["EPIC-KITCHENS-55"],"data_loaders":[],"num_papers_in_archive":42,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/egocentric-activity-recognition-on-epic-1","task":"Egocentric Activity Recognition","dataset_variant":"EPIC-KITCHENS-55","rows":7,"metrics":["Actions Top-1 (S1)","Actions Top-1 (S2)"],"first_row_in_archive_order":{"model":"DEEP-HAL with ODF+SDF (AssembleNet++)","paper":"/paper/hallucinating-statistical-moment-and-subspace","metrics":{"Actions Top-1 (S1)":"35.8","Actions Top-1 (S2)":"27.3"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-recognition-in-videos-on-epic-kitchens","task":"Action Recognition","dataset_variant":"EPIC-KITCHENS-55","rows":1,"metrics":["Top-1 Accuracy"],"first_row_in_archive_order":{"model":"TSM+W3 - full res","paper":"/paper/knowing-what-where-and-when-to-look-efficient","metrics":{"Top-1 Accuracy":"34.2"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-object-detection-on-epic-kitchens-55","task":"Video Object Detection","dataset_variant":"EPIC-KITCHENS-55","rows":1,"metrics":["mAP@.5"],"first_row_in_archive_order":{"model":"Ours (Faster RCNN)","paper":"/paper/objects-do-not-disappear-video-object","metrics":{"mAP@.5":"41.7"},"code_links":[{"title":"l-kid/video-object-detection-by-location-anticipation","url":"https://github.com/l-kid/video-object-detection-by-location-anticipation"},{"title":"Elstuhn/Video-object-detection-by-location-anticipation","url":"https://github.com/Elstuhn/Video-object-detection-by-location-anticipation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/objects-do-not-disappear-video-object","title":"Objects do not disappear: Video object detection by single-frame object location anticipation","date":"2023-08-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/knowing-what-where-and-when-to-look-efficient","title":"Knowing What, Where and When to Look: Efficient Video Action Modeling with Attention","date":"2020-04-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hallucinating-statistical-moment-and-subspace","title":"Self-supervising Action Recognition by Statistical Moment and Subspace Descriptors","date":"2020-01-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/epic-fusion-audio-visual-temporal-binding-for","title":"EPIC-Fusion: Audio-Visual Temporal Binding for Egocentric Action Recognition","date":"2019-08-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/what-would-you-expect-anticipating-egocentric","title":"What Would You Expect? Anticipating Egocentric Actions with Rolling-Unrolling LSTMs and Modality Attention","date":"2019-05-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/large-scale-weakly-supervised-pre-training","title":"Large-scale weakly-supervised pre-training for video action recognition","date":"2019-05-02","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":0,"samples_unverified":21,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/long-term-feature-banks-for-detailed-video","title":"Long-Term Feature Banks for Detailed Video Understanding","date":"2018-12-12","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lsta-long-short-term-attention-for-egocentric","title":"LSTA: Long Short-Term Attention for Egocentric Action Recognition","date":"2018-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":29,"samples_ran":0,"samples_unverified":29,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}