{"url":"/dataset/coco-captions","name":"COCO Captions","full_name":null,"description_markdown":"COCO Captions contains over one and a half million captions describing over 330,000 images. For the training and validation images, five independent human generated captions are be provided for each image.\r\n\r\nSource: [Microsoft COCO Captions: Data Collection and Evaluation Server](https://arxiv.org/abs/1504.00325)","description_withheld":null,"homepage":"https://github.com/tylin/coco-caption","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/microsoft-coco-captions-data-collection-and","title":"Microsoft COCO Captions: Data Collection and Evaluation Server","first_author":"Xinlei Chen","url":null},"license":{"name":"CC BY","url":"https://cocodataset.org/#termsofuse"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Concept-To-Text Generation","url":"/task/concept-to-text-generation","datasets_with_task":"/datasets/task/concept-to-text-generation"}],"languages":[],"variants":["COCO Captions","COCO Captions test","Image Captioning on COCO Captions","COCO Captions Karpathy Test"],"data_loaders":[{"repo":"https://github.com/pytorch/vision","url":"https://pytorch.org/vision/stable/generated/torchvision.datasets.CocoCaptions.html","frameworks":["pytorch"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#coco_captions","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/coco_captions","frameworks":["tf","jax"]},{"repo":"https://github.com/tylin/coco-caption","url":"https://github.com/tylin/coco-caption","frameworks":[]}],"num_papers_in_archive":203,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/image-captioning-on-coco-captions","task":"Image Captioning","dataset_variant":"COCO Captions","rows":41,"metrics":["BLEU-4","CIDER","METEOR","SPICE","ROUGE-L","BLEU-1","BLEU-2","BLEU-3","CLIPScore"],"first_row_in_archive_order":{"model":"mPLUG","paper":"/paper/mplug-effective-and-efficient-vision-language","metrics":{"BLEU-4":"46.5","CIDER":"155.1","METEOR":"32.0","SPICE":"26.0"},"code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"alibaba/AliceMind","url":"https://github.com/alibaba/AliceMind/tree/main/mPLUG"},{"title":"x-plug/mplug","url":"https://github.com/x-plug/mplug"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-coco-captions","task":"Text Generation","dataset_variant":"COCO Captions","rows":5,"metrics":["BLEU-2","BLEU-3","BLEU-4","BLEU-5"],"first_row_in_archive_order":{"model":"LeakGAN","paper":"/paper/long-text-generation-via-adversarial-training","metrics":{"BLEU-2":"0.950","BLEU-3":"0.880","BLEU-4":"0.778","BLEU-5":"0.686"},"code_links":[{"title":"CR-Gjx/LeakGAN","url":"https://github.com/CR-Gjx/LeakGAN"},{"title":"nurpeiis/LeakGAN-PyTorch","url":"https://github.com/nurpeiis/LeakGAN-PyTorch"},{"title":"valko073/LyricsGANs","url":"https://github.com/valko073/LyricsGANs"},{"title":"rupes438/CodeGen","url":"https://github.com/rupes438/CodeGen"},{"title":"liyzcj/leakgan-py3","url":"https://github.com/liyzcj/leakgan-py3"},{"title":"universebh/text_generation_fsa_gan","url":"https://github.com/universebh/text_generation_fsa_gan"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-coco-captions-test","task":"Image Captioning","dataset_variant":"COCO Captions test","rows":2,"metrics":["BLEU-4","CIDEr","METEOR","SPICE"],"first_row_in_archive_order":{"model":"From Captions to Visual Concepts and Back","paper":"/paper/from-captions-to-visual-concepts-and-back","metrics":{"BLEU-4":"56.7","CIDEr":"92.5","METEOR":"33.1"},"code_links":[{"title":"s-gupta/visual-concepts","url":"https://github.com/s-gupta/visual-concepts"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/concept-to-text-generation-on-coco-captions","task":"Concept-To-Text Generation","dataset_variant":"COCO Captions","rows":1,"metrics":["BLEU-2"],"first_row_in_archive_order":{"model":"tecpic","paper":"/paper/fake-news-detection-as-natural-language","metrics":{"BLEU-2":"2"},"code_links":[{"title":"zake7749/WSDM-Cup-2019","url":"https://github.com/zake7749/WSDM-Cup-2019"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/ladic-are-diffusion-models-really-inferior-to","title":"LaDiC: Are Diffusion Models Really Inferior to Autoregressive Counterparts for Image-to-Text Generation?","date":"2024-04-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fusecap-leveraging-large-language-models-to","title":"FuseCap: Leveraging Large Language Models for Enriched Fused Image Captions","date":"2023-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/prismer-a-vision-language-model-with-an","title":"Prismer: A Vision-Language Model with Multi-Task Experts","date":"2023-03-04","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","rows_on_this_dataset":3,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-only-training-for-image-captioning-using","title":"Text-Only Training for Image Captioning using Noise-Injected CLIP","date":"2022-11-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/smallcap-lightweight-image-captioning","title":"SmallCap: Lightweight Image Captioning Prompted with Retrieval Augmentation","date":"2022-09-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/expansionnet-v2-block-static-expansion-in","title":"Exploiting Multiple Sequence Lengths in Fast End to End Training for Image Captioning","date":"2022-08-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/prompt-tuning-for-generative-multimodal","title":"Prompt Tuning for Generative Multimodal Pretrained Models","date":"2022-08-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/grit-faster-and-better-image-captioning","title":"GRIT: Faster and Better Image captioning Transformer Using Dual Visual Features","date":"2022-07-20","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":0,"samples_unverified":21,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fine-grained-image-captioning-with-clip","title":"Fine-grained Image Captioning with CLIP Reward","date":"2022-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mplug-effective-and-efficient-vision-language","title":"mPLUG: Effective and Efficient Vision-Language Learning by Cross-modal Skip-connections","date":"2022-05-24","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/beyond-a-pre-trained-object-detector-cross","title":"Beyond a Pre-Trained Object Detector: Cross-Modal Textual and Visual Context for Image Captioning","date":"2022-05-09","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-up-vision-language-pre-training-for","title":"Scaling Up Vision-Language Pre-training for Image Captioning","date":"2021-11-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/l-verse-bidirectional-generation-between","title":"L-Verse: Bidirectional Generation Between Image and Text","date":"2021-11-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/clipcap-clip-prefix-for-image-captioning","title":"ClipCap: CLIP Prefix for Image Captioning","date":"2021-11-18","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/enabling-multimodal-generation-on-clip-via","title":"Enabling Multimodal Generation on CLIP via Vision-Language Knowledge Distillation","date":"2021-11-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/refinecap-concept-aware-refinement-for-image","title":"RefineCap: Concept-Aware Refinement for Image Captioning","date":"2021-09-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/simvlm-simple-visual-language-model","title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","date":"2021-08-24","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":18,"samples_unverified":19,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vinvl-making-visual-representations-matter-in","title":"VinVL: Revisiting Visual Representations in Vision-Language Models","date":"2021-01-02","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/virtex-learning-visual-representations-from","title":"VirTex: Learning Visual Representations from Textual Annotations","date":"2020-06-11","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/oscar-object-semantics-aligned-pre-training","title":"Oscar: Object-Semantics Aligned Pre-training for Vision-Language Tasks","date":"2020-04-13","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":11,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/x-linear-attention-networks-for-image","title":"X-Linear Attention Networks for Image Captioning","date":"2020-03-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-better-variant-of-self-critical-sequence","title":"A Better Variant of Self-Critical Sequence Training","date":"2020-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/visual-commonsense-r-cnn","title":"Visual Commonsense R-CNN","date":"2020-02-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/m2-meshed-memory-transformer-for-image","title":"Meshed-Memory Transformer for Image Captioning","date":"2019-12-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unified-vision-language-pre-training-for","title":"Unified Vision-Language Pre-Training for Image Captioning and VQA","date":"2019-09-24","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reflective-decoding-network-for-image","title":"Reflective Decoding Network for Image Captioning","date":"2019-08-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fake-news-detection-as-natural-language","title":"Fake News Detection as Natural Language Inference","date":"2019-07-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/relgan-relational-generative-adversarial","title":"RelGAN: Relational Generative Adversarial Networks for Text Generation","date":"2019-05-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/long-text-generation-via-adversarial-training","title":"Long Text Generation via Adversarial Training with Leaked Information","date":"2017-09-24","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adversarial-ranking-for-language-generation","title":"Adversarial Ranking for Language Generation","date":"2017-05-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/seqgan-sequence-generative-adversarial-nets","title":"SeqGAN: Sequence Generative Adversarial Nets with Policy Gradient","date":"2016-09-18","rows_on_this_dataset":1,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":10,"samples_unverified":11,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/from-captions-to-visual-concepts-and-back","title":"From Captions to Visual Concepts and Back","date":"2014-11-18","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":20,"samples_harvested":225,"samples_ran":94,"samples_unverified":131,"pointer_only_for_licence":60,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}