{"url":"/dataset/youcook2","name":"YouCook2","full_name":null,"description_markdown":"**YouCook2** is the largest task-oriented, instructional video dataset in the vision community. It contains 2000 long untrimmed videos from 89 cooking recipes; on average, each distinct recipe has 22 videos. The procedure steps for each video are annotated with temporal boundaries and described by imperative English sentences (see the example below). The videos were downloaded from YouTube and are all in the third-person viewpoint. All the videos are unconstrained and can be performed by individual persons at their houses with unfixed cameras. YouCook2 contains rich recipe types and various cooking styles from all over the world.\r\n\r\nSource: [http://youcook2.eecs.umich.edu/](http://youcook2.eecs.umich.edu/)\r\nImage Source: [https://competitions.codalab.org/competitions/20594](https://competitions.codalab.org/competitions/20594)","description_withheld":null,"homepage":"http://youcook2.eecs.umich.edu/","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/towards-automatic-learning-of-procedures-from","title":"Towards Automatic Learning of Procedures from Web Instructional Videos","first_author":"Luowei Zhou","url":null},"license":{"name":"Custom","url":"http://youcook2.eecs.umich.edu/download"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Action Classification","url":"/task/action-classification","datasets_with_task":"/datasets/task/action-classification"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"},{"name":"Zero-Shot Video Retrieval","url":"/task/zero-shot-video-retrieval","datasets_with_task":"/datasets/task/zero-shot-video-retrieval"},{"name":"Dense Video Captioning","url":"/task/dense-video-captioning","datasets_with_task":"/datasets/task/dense-video-captioning"},{"name":"Zero-shot dense video captioning","url":"/task/zero-shot-dense-video-captioning","datasets_with_task":"/datasets/task/zero-shot-dense-video-captioning"},{"name":"Zero-Shot Video-Audio Retrieval","url":"/task/zero-shot-video-audio-retrieval","datasets_with_task":"/datasets/task/zero-shot-video-audio-retrieval"},{"name":"Long Video Retrieval (Background Removed)","url":"/task/long-video-retrieval-background-removed","datasets_with_task":"/datasets/task/long-video-retrieval-background-removed"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["YouCook2"],"data_loaders":[],"num_papers_in_archive":198,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-retrieval-on-youcook2","task":"Video Retrieval","dataset_variant":"YouCook2","rows":16,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video Median Rank","text-to-video Mean Rank"],"first_row_in_archive_order":{"model":"VAST","paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","metrics":{"text-to-video R@1":"50.4","text-to-video R@10":"80.8","text-to-video R@5":"74.3"},"code_links":[{"title":"TXH-mercury/VALOR","url":"https://github.com/TXH-mercury/VALOR"},{"title":"txh-mercury/vast","url":"https://github.com/txh-mercury/vast"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-captioning-on-youcook2","task":"Video Captioning","dataset_variant":"YouCook2","rows":14,"metrics":["BLEU-4","BLEU-3","CIDEr","ROUGE-L","METEOR"],"first_row_in_archive_order":{"model":"VAST","paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","metrics":{"BLEU-4":"18.2","CIDEr":"1.99"},"code_links":[{"title":"TXH-mercury/VALOR","url":"https://github.com/TXH-mercury/VALOR"},{"title":"txh-mercury/vast","url":"https://github.com/txh-mercury/vast"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-retrieval-on-youcook2","task":"Zero-Shot Video Retrieval","dataset_variant":"YouCook2","rows":9,"metrics":["text-to-video R@1","text-to-video R@5","text-to-video R@10","text-to-video Mean Rank","text-to-video Median Rank"],"first_row_in_archive_order":{"model":"OmniVec2","paper":"/paper/omnivec2-a-novel-transformer-based-network","metrics":{"text-to-video R@1":"26.1","text-to-video R@10":"70.8","text-to-video R@5":"54.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dense-video-captioning-on-youcook2","task":"Dense Video Captioning","dataset_variant":"YouCook2","rows":7,"metrics":["CIDEr","METEOR","SODA","BLEU4","ROUGE-L","F1","Precision","Recall"],"first_row_in_archive_order":{"model":"HiCM²","paper":"/paper/hicm-2-hierarchical-compact-memory-modeling","metrics":{"BLEU4":"6.11","CIDEr":"71.84","F1":"32.51","METEOR":"12.80","Precision":"32.51","Recall":"32.51","SODA":"10.73"},"code_links":[{"title":"ailab-kyunghee/HiCM2-DVC","url":"https://github.com/ailab-kyunghee/HiCM2-DVC"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/long-video-retrieval-background-removed-on","task":"Long Video Retrieval (Background Removed)","dataset_variant":"YouCook2","rows":6,"metrics":["Cap. Avg. R@1","Cap. Avg. R@5","Cap. Avg. R@10","DTW R@1","DTW R@5","DTW R@10","OTAM R@1","OTAM R@5","OTAM R@10"],"first_row_in_archive_order":{"model":"Norton","paper":"/paper/multi-granularity-correspondence-learning-1","metrics":{"Cap. Avg. R@1":"75.5","Cap. Avg. R@10":"97.7","Cap. Avg. R@5":"95.0","DTW R@1":"88.7","DTW R@10":"99.5","DTW R@5":"98.8","OTAM R@1":"88.9","OTAM R@10":"99.5","OTAM R@5":"98.4"},"code_links":[{"title":"XLearning-SCU/2024-ICLR-Norton","url":"https://github.com/XLearning-SCU/2024-ICLR-Norton"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/action-classification-on-youcook2","task":"Action Classification","dataset_variant":"YouCook2","rows":1,"metrics":["Object Top 5 Accuracy","Object Top-1 Accuracy","Verb Top-1 Accuracy","Verb Top-5 Accuracy"],"first_row_in_archive_order":{"model":"VideoBERT (cross modal)","paper":"/paper/videobert-a-joint-model-for-video-and","metrics":{"Object Top 5 Accuracy":"33.7","Object Top-1 Accuracy":"13.1","Verb Top-1 Accuracy":"3.2","Verb Top-5 Accuracy":"43.3"},"code_links":[{"title":"ammesatyajit/VideoBERT","url":"https://github.com/ammesatyajit/VideoBERT"},{"title":"MDSKUL/MasterProject","url":"https://github.com/MDSKUL/MasterProject"},{"title":"parkervg/allrecipes-bert","url":"https://github.com/parkervg/allrecipes-bert"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-dense-video-captioning-on-youcook2","task":"Zero-shot dense video captioning","dataset_variant":"YouCook2","rows":1,"metrics":["CIDEr","METEOR","SODA"],"first_row_in_archive_order":{"model":"Vid2Seq (HowTo100M+VidChapters-7M PT)","paper":null,"metrics":{"CIDEr":"13.3","METEOR":"3.4","SODA":"3.9"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/hicm-2-hierarchical-compact-memory-modeling","title":"HiCM$^2$: Hierarchical Compact Memory Modeling for Dense Video Captioning","date":"2024-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/do-you-remember-dense-video-captioning-with","title":"Do You Remember? Dense Video Captioning with Cross-Modal Memory Retrieval","date":"2024-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-granularity-correspondence-learning-1","title":"Multi-granularity Correspondence Learning from Long-term Noisy Videos","date":"2024-01-30","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivec2-a-novel-transformer-based-network","title":"OmniVec2 - A Novel Transformer based Network for Large Scale Multimodal and Multitask Learning","date":"2024-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/omnivec-learning-robust-representations-with","title":"OmniVec: Learning robust representations with cross modal sharing","date":"2023-11-07","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-with-knowledge-graph-augmented","title":"Text with Knowledge Graph Augmented Transformer for Video Captioning","date":"2023-03-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-grounded-vision-language","title":"Learning Grounded Vision-Language Representation for Versatile Understanding in Untrimmed Videos","date":"2023-03-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":1,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vid2seq-large-scale-pretraining-of-a-visual","title":"Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense Video Captioning","date":"2023-02-27","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/tempclr-temporal-alignment-representation","title":"TempCLR: Temporal Alignment Representation with Contrastive Learning","date":"2022-12-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/semantic-role-aware-correlation-transformer","title":"Semantic Role Aware Correlation Transformer for Text to Video Retrieval","date":"2022-06-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rome-role-aware-mixture-of-expert-transformer","title":"RoME: Role-aware Mixture-of-Expert Transformer for Text-to-Video Retrieval","date":"2022-06-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mdmmt-2-multidomain-multimodal-transformer","title":"MDMMT-2: Multidomain Multimodal Transformer for Video Retrieval, One More Step Towards Generalization","date":"2022-03-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/videoclip-contrastive-pre-training-for-zero","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","date":"2021-09-28","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/taco-token-aware-cascade-contrastive-learning","title":"TACo: Token-aware Cascade Contrastive Learning for Video-Text Alignment","date":"2021-08-23","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-dense-video-captioning-with","title":"End-to-End Dense Video Captioning with Parallel Decoding","date":"2021-08-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vlm-task-agnostic-video-language-model-pre","title":"VLM: Task-agnostic Video-Language Model Pre-training for Video Understanding","date":"2021-05-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multimodal-clustering-networks-for-self","title":"Multimodal Clustering Networks for Self-supervised Learning from Unlabeled Videos","date":"2021-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vatt-transformers-for-multimodal-self","title":"VATT: Transformers for Multimodal Self-Supervised Learning from Raw Video, Audio and Text","date":"2021-04-22","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-pretraining-for-dense-video","title":"Multimodal Pretraining for Dense Video Captioning","date":"2020-11-10","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/univilm-a-unified-video-and-language-pre","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","date":"2020-02-15","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-learning-of-visual-representations","title":"End-to-End Learning of Visual Representations from Uncurated Instructional Videos","date":"2019-12-13","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/howto100m-learning-a-text-video-embedding-by","title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","date":"2019-06-07","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videobert-a-joint-model-for-video-and","title":"VideoBERT: A Joint Model for Video and Language Representation Learning","date":"2019-04-03","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-dense-video-captioning-with-masked","title":"End-to-End Dense Video Captioning with Masked Transformer","date":"2018-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/associating-neural-word-embeddings-with-deep","title":"Associating Neural Word Embeddings With Deep Image Representations Using Fisher Vectors","date":"2015-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":16,"samples_harvested":113,"samples_ran":49,"samples_unverified":64,"pointer_only_for_licence":16,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}