{"url":"/dataset/breakfast","name":"Breakfast","full_name":"The Breakfast Actions Dataset","description_markdown":"The **Breakfast** Actions Dataset comprises of 10 actions related to breakfast preparation, performed by 52 different individuals in 18 different kitchens. The dataset is one of the largest fully annotated datasets available. The actions are recorded “in the wild” as opposed to a single controlled lab environment. It consists of over 77 hours of video recordings.\r\n\r\nSource: [https://serre-lab.clps.brown.edu/resource/breakfast-actions-dataset/](https://serre-lab.clps.brown.edu/resource/breakfast-actions-dataset/)\r\nImage Source: [https://serre-lab.clps.brown.edu/resource/breakfast-actions-dataset/](https://serre-lab.clps.brown.edu/resource/breakfast-actions-dataset/)","description_withheld":null,"homepage":"https://serre-lab.clps.brown.edu/resource/breakfast-actions-dataset/","introduced_date":"2014-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-language-of-actions-recovering-the-syntax","title":"The Language of Actions: Recovering the Syntax and Semantics of Goal-Directed Human Activities","first_author":"Hilde Kuehne","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Actions","url":"/datasets/modality/actions"}],"tasks":[{"name":"Action Segmentation","url":"/task/action-segmentation","datasets_with_task":"/datasets/task/action-segmentation"},{"name":"Video Classification","url":"/task/video-classification","datasets_with_task":"/datasets/task/video-classification"},{"name":"Unsupervised Action Segmentation","url":"/task/unsupervised-action-segmentation","datasets_with_task":"/datasets/task/unsupervised-action-segmentation"},{"name":"Weakly Supervised Action Segmentation (Transcript)","url":"/task/weakly-supervised-action-segmentation","datasets_with_task":"/datasets/task/weakly-supervised-action-segmentation"},{"name":"Weakly Supervised Action Segmentation (Action Set))","url":"/task/weakly-supervised-action-segmentation-action","datasets_with_task":"/datasets/task/weakly-supervised-action-segmentation-action"},{"name":"Long-video Activity Recognition","url":"/task/long-video-activity-recognition","datasets_with_task":"/datasets/task/long-video-activity-recognition"}],"languages":[],"variants":["Breakfast"],"data_loaders":[],"num_papers_in_archive":179,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-segmentation-on-breakfast-1","task":"Action Segmentation","dataset_variant":"Breakfast","rows":37,"metrics":["Average F1","F1@50%","F1@25%","F1@10%","Edit","Acc","mIoU","F1"],"first_row_in_archive_order":{"model":"AdaFocus (newly extracted I3D-features, LT-Context model)","paper":"/paper/adafocus-towards-end-to-end-weakly-supervised","metrics":{"Acc":"78.0","Average F1":"76.2","Edit":"78.3","F1@10%":"82.1","F1@25%":"79.0","F1@50%":"67.5"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-classification-on-breakfast","task":"Video Classification","dataset_variant":"Breakfast","rows":9,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"HERMES","paper":"/paper/bridging-episodes-and-semantics-a-novel","metrics":{"Accuracy (%)":"95.2"},"code_links":[{"title":"joslefaure/HERMES","url":"https://github.com/joslefaure/HERMES"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/long-video-activity-recognition-on-breakfast","task":"Long-video Activity Recognition","dataset_variant":"Breakfast","rows":8,"metrics":["mAP"],"first_row_in_archive_order":{"model":"AdaFocus (MViT-Breakfast-Pretrain-feature, GHRM)","paper":"/paper/adafocus-towards-end-to-end-weakly-supervised","metrics":{"mAP":"79.5"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-action-segmentation-on-breakfast","task":"Unsupervised Action Segmentation","dataset_variant":"Breakfast","rows":8,"metrics":["F1","Acc","JSD","Precision","Recall","mIoU"],"first_row_in_archive_order":{"model":"HVQ","paper":"/paper/hierarchical-vector-quantization-for","metrics":{"Acc":"54.4","F1":"39.7","JSD":"82.5","Precision":"35.6","Recall":"44.9"},"code_links":[{"title":"fedespu/hvq","url":"https://github.com/fedespu/hvq"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-action-segmentation","task":"Weakly Supervised Action Segmentation (Transcript)","dataset_variant":"Breakfast","rows":7,"metrics":["Acc"],"first_row_in_archive_order":{"model":"AUL","paper":"/paper/is-weakly-supervised-action-segmentation","metrics":{"Acc":"67.3"},"code_links":[{"title":"fandulu/DD-Net","url":"https://github.com/fandulu/DD-Net"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-action-segmentation-action","task":"Weakly Supervised Action Segmentation (Action Set))","dataset_variant":"Breakfast","rows":2,"metrics":["Acc"],"first_row_in_archive_order":{"model":"AdaFocus (newly extracted I3D-features, POC model)","paper":"/paper/adafocus-towards-end-to-end-weakly-supervised","metrics":{"Acc":"49.6"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hierarchical-vector-quantization-for","title":"Hierarchical Vector Quantization for Unsupervised Action Segmentation","date":"2024-12-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/asquery-a-query-based-model-for-action","title":"ASQuery: A Query-based Model for Action Segmentation","date":"2024-09-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bridging-episodes-and-semantics-a-novel","title":"HERMES: temporal-coHERent long-forM understanding with Episodes and Semantics","date":"2024-08-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-temporal-action-segmentation-via","title":"Efficient Temporal Action Segmentation via Boundary-aware Query Voting","date":"2024-05-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporally-consistent-unbalanced-optimal","title":"Temporally Consistent Unbalanced Optimal Transport for Unsupervised Action Segmentation","date":"2024-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fact-frame-action-cross-attention-temporal","title":"FACT: Frame-Action Cross-Attention Temporal Modeling for Efficient Action Segmentation","date":"2024-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adafocus-towards-end-to-end-weakly-supervised","title":"Towards Weakly Supervised End-to-end Learning for Long-video Action Recognition","date":"2023-11-28","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/is-weakly-supervised-action-segmentation","title":"Is Weakly-supervised Action Segmentation Ready For Human-Robot Interaction? No, Let's Improve It With Action-union Learning","date":"2023-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bit-bi-level-temporal-modeling-for-efficient","title":"BIT: Bi-Level Temporal Modeling for Efficient Supervised Action Segmentation","date":"2023-08-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/how-much-temporal-long-term-context-is-needed","title":"How Much Temporal Long-Term Context is Needed for Action Segmentation?","date":"2023-08-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sf-tmn-slowfast-temporal-modeling-network-for","title":"SF-TMN: SlowFast Temporal Modeling Network for Surgical Phase Recognition","date":"2023-06-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/permutation-aware-action-segmentation-via","title":"Permutation-Aware Action Segmentation via Unsupervised Frame-to-Segment Alignment","date":"2023-05-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/leveraging-triplet-loss-for-unsupervised","title":"Leveraging triplet loss for unsupervised action segmentation","date":"2023-04-13","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/diffusion-action-segmentation","title":"Diffusion Action Segmentation","date":"2023-03-31","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":5,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/selective-structured-state-spaces-for-long","title":"Selective Structured State-Spaces for Long-Form Video Understanding","date":"2023-03-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/aspnet-action-segmentation-with-shared","title":"ASPnet: Action Segmentation With Shared-Private Representation of Multiple Data Sources","date":"2023-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-movie-scene-detection-using-state","title":"Efficient Movie Scene Detection using State-Space Transformers","date":"2022-12-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unified-fully-and-timestamp-supervised","title":"Unified Fully and Timestamp Supervised Temporal Action Segmentation via Sequence to Sequence Translation","date":"2022-09-01","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rf-next-efficient-receptive-field-search-for","title":"RF-Next: Efficient Receptive Field Search for Convolutional Neural Networks","date":"2022-06-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-u-transformer-with-boundary-aware","title":"Do we really need temporal convolutions in action segmentation?","date":"2022-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cross-enhancement-transformer-for-action","title":"Cross-Enhancement Transformer for Action Segmentation","date":"2022-05-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/maximization-and-restoration-action","title":"Maximization and restoration: Action segmentation through dilation passing and temporal reconstruction","date":"2022-05-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/long-movie-clip-classification-with-state","title":"Long Movie Clip Classification with State-Space Video Models","date":"2022-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":11,"samples_unverified":7,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-recognize-procedural-activities","title":"Learning To Recognize Procedural Activities with Distant Supervision","date":"2022-01-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/set-supervised-action-learning-in-procedural","title":"Set-Supervised Action Learning in Procedural Task Videos via Pairwise Order Consistency","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/asformer-transformer-for-action-segmentation","title":"ASFormer: Transformer for Action Segmentation","date":"2021-10-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fifa-fast-inference-approximation-for-action","title":"FIFA: Fast Inference Approximation for Action Segmentation","date":"2021-08-09","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/graph-based-high-order-relation-modeling-for","title":"Graph-Based High-Order Relation Modeling for Long-Term Action Recognition","date":"2021-06-19","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/unsupervised-activity-segmentation-by-joint","title":"Unsupervised Action Segmentation by Joint Representation Learning and Online Clustering","date":"2021-05-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coarse-to-fine-multi-resolution-temporal","title":"Coarse to Fine Multi-Resolution Temporal Convolutional Network","date":"2021-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unsupervised-discriminative-embedding-for-sub","title":"Unsupervised Discriminative Embedding for Sub-Action Learning in Complex Activities","date":"2021-04-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-two-step-networks-for-temporal","title":"Efficient Two-Step Networks for Temporal Action Segmentation","date":"2021-04-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/action-segmentation-with-mixed-temporal","title":"Action Segmentation with Mixed Temporal Domain Adaptation","date":"2021-04-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/action-shuffle-alternating-learning-for","title":"Action Shuffle Alternating Learning for Unsupervised Action Segmentation","date":"2021-04-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporally-weighted-hierarchical-clustering","title":"Temporally-Weighted Hierarchical Clustering for Unsupervised Action Segmentation","date":"2021-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/depthwise-separable-temporal-convolutional","title":"Depthwise Separable Temporal Convolutional Network for Action Segmentation","date":"2021-01-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/global2local-efficient-structure-search-for","title":"Global2Local: Efficient Structure Search for Video Action Segmentation","date":"2021-01-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weakly-supervised-action-segmentation-and","title":"Weakly-Supervised Action Segmentation and Alignment via Transcript-Aware Union-of-Subspaces Learning","date":"2021-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/refining-action-segmentation-with","title":"Refining Action Segmentation With Hierarchical Video Representations","date":"2021-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/temporal-relational-modeling-with-self","title":"Temporal Relational Modeling with Self-Supervision for Action Segmentation","date":"2020-12-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/boundary-aware-cascade-networks-for-temporal","title":"Boundary-Aware Cascade Networks for Temporal Action Segmentation","date":"2020-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/alleviating-over-segmentation-errors-by","title":"Alleviating Over-segmentation Errors by Detecting Action Boundaries","date":"2020-07-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ms-tcn-multi-stage-temporal-convolutional-2","title":"MS-TCN++: Multi-Stage Temporal Convolutional Network for Action Segmentation","date":"2020-06-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/improving-action-segmentation-via-graph-based","title":"Improving Action Segmentation via Graph-Based Temporal Reasoning","date":"2020-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/action-segmentation-with-joint-self","title":"Action Segmentation with Joint Self-Supervised Temporal Domain Adaptation","date":"2020-03-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weakly-supervised-energy-based-learning-for","title":"Weakly Supervised Energy-Based Learning for Action Segmentation","date":"2019-09-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/videograph-recognizing-minutes-long-human","title":"VideoGraph: Recognizing Minutes-Long Human Activities in Videos","date":"2019-05-13","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/unsupervised-learning-of-action-classes-with","title":"Unsupervised learning of action classes with continuous temporal embedding","date":"2019-04-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/weakly-supervised-action-segmentation-using","title":"Fast Weakly Supervised Action Segmentation Using Mutual Consistency","date":"2019-04-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ms-tcn-multi-stage-temporal-convolutional","title":"MS-TCN: Multi-Stage Temporal Convolutional Network for Action Segmentation","date":"2019-03-05","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/d3tw-discriminative-differentiable-dynamic","title":"D3TW: Discriminative Differentiable Dynamic Time Warping for Weakly Supervised Action Alignment and Segmentation","date":"2019-01-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/timeception-for-complex-action-recognition","title":"Timeception for Complex Action Recognition","date":"2018-12-04","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/a-perceptual-prediction-framework-for-self","title":"A Perceptual Prediction Framework for Self Supervised Event Segmentation","date":"2018-11-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neuralnetwork-viterbi-a-framework-for-weakly","title":"NeuralNetwork-Viterbi: A Framework for Weakly Supervised Video Learning","date":"2018-05-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/actionvlad-learning-spatio-temporal","title":"ActionVLAD: Learning spatio-temporal aggregation for action classification","date":"2017-04-10","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":15,"samples_harvested":87,"samples_ran":55,"samples_unverified":32,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}