{"url":"/dataset/conceptual-captions","name":"Conceptual Captions","full_name":"Conceptual Captions","description_markdown":"Automatic image captioning is the task of producing a natural-language utterance (usually a sentence) that correctly reflects the visual content of an image. Up to this point, the resource most used for this task was the MS-COCO dataset, containing around 120,000 images and 5-way image-caption annotations (produced by paid annotators).\r\n\r\nGoogle's Conceptual Captions dataset has more than 3 million images, paired with natural-language captions. In contrast with the curated style of the MS-COCO images, Conceptual Captions images and their raw descriptions are harvested from the web, and therefore represent a wider variety of styles. The raw descriptions are harvested from the Alt-text HTML attribute associated with web images. The authors developed an automatic pipeline that extracts, filters, and transforms candidate image/caption pairs, with the goal of achieving a balance of cleanliness, informativeness, fluency, and learnability of the resulting captions.\r\n\r\nSource: [Conceptual Captions](https://github.com/google-research-datasets/conceptual-captions)\r\nImage Source: [Sharma et al](https://www.aclweb.org/anthology/P18-1238)","description_withheld":null,"homepage":"https://github.com/google-research-datasets/conceptual-captions","introduced_date":"2018-07-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/conceptual-captions-a-cleaned-hypernymed","title":"Conceptual Captions: A Cleaned, Hypernymed, Image Alt-text Dataset For Automatic Image Captioning","first_author":"Piyush Sharma","url":null},"license":{"name":"Custom","url":"https://github.com/google-research-datasets/conceptual-captions/blob/master/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Text-to-Image Generation","url":"/task/text-to-image-generation","datasets_with_task":"/datasets/task/text-to-image-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Conceptual Captions"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/google-research-datasets/conceptual_captions","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/conceptual_captions","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/google-research-datasets/conceptual-captions","url":"https://github.com/google-research-datasets/conceptual-captions","frameworks":[]}],"num_papers_in_archive":352,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-to-image-generation-on-conceptual","task":"Text-to-Image Generation","dataset_variant":"Conceptual Captions","rows":5,"metrics":["FID"],"first_row_in_archive_order":{"model":"Contextual RQ-Transformer","paper":"/paper/draft-and-revise-effective-image-generation","metrics":{"FID":"9.80"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-conceptual-captions","task":"Image Captioning","dataset_variant":"Conceptual Captions","rows":2,"metrics":["CIDEr","ROUGE-L","SPICE"],"first_row_in_archive_order":{"model":"ClipCap (MLP + GPT2 tuning)","paper":"/paper/clipcap-clip-prefix-for-image-captioning","metrics":{"CIDEr":"87.26","ROUGE-L":"26.71","SPICE":"18.5"},"code_links":[{"title":"rmokady/clip_prefix_caption","url":"https://github.com/rmokady/clip_prefix_caption"},{"title":"Japanese-Image-Captioning/ClipCap-for-Japanese","url":"https://github.com/Japanese-Image-Captioning/ClipCap-for-Japanese"},{"title":"sithu31296/image-captioning","url":"https://github.com/sithu31296/image-captioning"},{"title":"MS-P3/code7","url":"https://github.com/MS-P3/code7/tree/main/x_clip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/draft-and-revise-effective-image-generation","title":"Draft-and-Revise: Effective Image Generation with Contextual RQ-Transformer","date":"2022-06-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/autoregressive-image-generation-using","title":"Autoregressive Image Generation using Residual Quantization","date":"2022-03-03","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/high-resolution-image-synthesis-with-latent","title":"High-Resolution Image Synthesis with Latent Diffusion Models","date":"2021-12-20","rows_on_this_dataset":1,"code_links":41,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":28,"samples_ran":19,"samples_unverified":9,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/clipcap-clip-prefix-for-image-captioning","title":"ClipCap: CLIP Prefix for Image Captioning","date":"2021-11-18","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/imagebart-bidirectional-context-with","title":"ImageBART: Bidirectional Context with Multinomial Diffusion for Autoregressive Image Synthesis","date":"2021-08-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/taming-transformers-for-high-resolution-image","title":"Taming Transformers for High-Resolution Image Synthesis","date":"2020-12-17","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":41,"samples_ran":29,"samples_unverified":12,"pointer_only_for_licence":10,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}