{"url":"/dataset/image-paragraph-captioning","name":"Image Paragraph Captioning","full_name":null,"description_markdown":"The Image Paragraph Captioning dataset allows researchers to benchmark their progress in generating paragraphs that tell a story about an image. The dataset contains 19,561 images from the [Visual Genome dataset](https://paperswithcode.com/dataset/visual-genome). Each image contains one paragraph. The training/val/test sets contains 14,575/2,487/2,489 images.\r\n\r\nSince all the images are also part of the Visual Genome dataset, each image also contains 50 region descriptions (short phrases describing parts of an image), 35 objects, 26 attributes and 21 relationships and 17 question-answer pairs.\r\n\r\nSource: [A Hierarchical Approach for Generating Descriptive Image Paragraphs](/paper/a-hierarchical-approach-for-generating)\r\nImage Source: [https://cs.stanford.edu/people/ranjaykrishna/im2p/index.html](https://cs.stanford.edu/people/ranjaykrishna/im2p/index.html)","description_withheld":null,"homepage":"https://cs.stanford.edu/people/ranjaykrishna/im2p/index.html","introduced_date":"2016-11-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-hierarchical-approach-for-generating","title":"A Hierarchical Approach for Generating Descriptive Image Paragraphs","first_author":"Jonathan Krause","url":null},"license":{"name":"Custom","url":"https://cs.stanford.edu/people/ranjaykrishna/im2p/index.html"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Image Paragraph Captioning","url":"/task/image-paragraph-captioning","datasets_with_task":"/datasets/task/image-paragraph-captioning"},{"name":"Zero-Shot Image Paragraph Captioning","url":"/task/zero-shot-image-paragraph-captioning","datasets_with_task":"/datasets/task/zero-shot-image-paragraph-captioning"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Image Paragraph Captioning"],"data_loaders":[],"num_papers_in_archive":31,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/image-paragraph-captioning-on-image-paragraph","task":"Image Paragraph Captioning","dataset_variant":"Image Paragraph Captioning","rows":10,"metrics":["BLEU-4","METEOR","CIDEr"],"first_row_in_archive_order":{"model":"HSGED(SLL)","paper":"/paper/hierarchical-scene-graph-encoder-decoder-for","metrics":{"BLEU-4":"11.26","CIDEr":"36.02","METEOR":"18.33"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/when-an-image-tells-a-story-the-role-of","title":"When an Image Tells a Story: The Role of Visual and Semantic Information for Generating Paragraph Descriptions","date":"2020-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/interactive-key-value-memory-augmented","title":"Interactive Key-Value Memory-augmented Attention for Image Paragraph Captioning","date":"2020-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hierarchical-scene-graph-encoder-decoder-for","title":"Hierarchical Scene Graph Encoder-Decoder for Image Paragraph Captioning","date":"2020-10-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dual-cnn-a-convolutional-language-decoder-for","title":"Dual-CNN: A Convolutional language decoder for paragraph image captioning","date":"2020-02-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/convolutional-auto-encoding-of-sentence","title":"Convolutional Auto-encoding of Sentence Topics for Image Paragraph Generation","date":"2019-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/look-deeper-see-richer-depth-aware-image","title":"Look Deeper See Richer: Depth-aware Image Paragraph Captioning","date":"2018-10-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/training-for-diversity-in-image-paragraph","title":"Training for Diversity in Image Paragraph Captioning","date":"2018-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/diverse-and-coherent-paragraph-generation","title":"Diverse and Coherent Paragraph Generation from Images","date":"2018-09-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/recurrent-topic-transition-gan-for-visual","title":"Recurrent Topic-Transition GAN for Visual Paragraph Generation","date":"2017-03-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-hierarchical-approach-for-generating","title":"A Hierarchical Approach for Generating Descriptive Image Paragraphs","date":"2016-11-20","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}