{"url":"/dataset/recipeqa","name":"RecipeQA","full_name":null,"description_markdown":"RecipeQA is a dataset for multimodal comprehension of cooking recipes. It consists of over 36K question-answer pairs automatically generated from approximately 20K unique recipes with step-by-step instructions and images. Each question in RecipeQA involves multiple modalities such as titles, descriptions or images, and working towards an answer requires (i) joint understanding of images and text, (ii) capturing the temporal flow of events, and (iii) making sense of procedural knowledge.\r\n\r\nSource: [RecipeQA](https://hucvl.github.io/recipeqa/)","description_withheld":null,"homepage":"https://hucvl.github.io/recipeqa/","introduced_date":"2018-09-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/recipeqa-a-challenge-dataset-for-multimodal","title":"RecipeQA: A Challenge Dataset for Multimodal Comprehension of Cooking Recipes","first_author":"Semih Yagcioglu","url":null},"license":{"name":"Custom","url":"https://hucvl.github.io/recipeqa/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Common Sense Reasoning","url":"/task/common-sense-reasoning","datasets_with_task":"/datasets/task/common-sense-reasoning"},{"name":"Natural Language Understanding","url":"/task/natural-language-understanding","datasets_with_task":"/datasets/task/natural-language-understanding"}],"languages":[],"variants":["RecipeQA"],"data_loaders":[],"num_papers_in_archive":24,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-recipeqa","task":"Question Answering","dataset_variant":"RecipeQA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"multimodal+LXMERT+ConstrainedMaxPooling","paper":"/paper/latent-alignment-of-procedural-concepts-in-1","metrics":{"Accuracy":"0.475"},"code_links":[{"title":"HLR/LatentAlignmentProcedural","url":"https://github.com/HLR/LatentAlignmentProcedural"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/latent-alignment-of-procedural-concepts-in-1","title":"Latent Alignment of Procedural Concepts in Multimodal Recipes","date":"2021-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}