{"url":"/dataset/flickr30k-entities","name":"Flickr30K Entities","full_name":null,"description_markdown":"The **Flickr30K Entities** dataset is an extension to the Flickr30K dataset. It augments the original 158k captions with 244k coreference chains, linking mentions of the same entities across different captions for the same image, and associating them with 276k manually annotated bounding boxes. This is used to define a new benchmark for localization of textual entity mentions in an image.\r\n\r\nSource: [http://bryanplummer.com/Flickr30kEntities/](http://bryanplummer.com/Flickr30kEntities/)\r\nImage Source: [http://bryanplummer.com/Flickr30kEntities/](http://bryanplummer.com/Flickr30kEntities/)","description_withheld":null,"homepage":"http://bryanplummer.com/Flickr30kEntities/","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/flickr30k-entities-collecting-region-to","title":"Flickr30k Entities: Collecting Region-to-Phrase Correspondences for Richer Image-to-Sentence Models","first_author":"Bryan A. Plummer","url":null},"license":{"name":"Custom (research-only, non-commercial)","url":"https://github.com/BryanPlummer/flickr30k_entities"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Phrase Grounding","url":"/task/phrase-grounding","datasets_with_task":"/datasets/task/phrase-grounding"}],"languages":[],"variants":["Flickr30k Entities Dev","Flickr30k Entities Test","Flickr30K Entities"],"data_loaders":[],"num_papers_in_archive":142,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/phrase-grounding-on-flickr30k-entities-test","task":"Phrase Grounding","dataset_variant":"Flickr30k Entities Test","rows":18,"metrics":["R@1","R@5","R@10"],"first_row_in_archive_order":{"model":"GLIPv2","paper":"/paper/glipv2-unifying-localization-and-vision","metrics":{"R@1":"87.7"},"code_links":[{"title":"microsoft/GLIP","url":"https://github.com/microsoft/GLIP"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/phrase-grounding-on-flickr30k-entities-dev","task":"Phrase Grounding","dataset_variant":"Flickr30k Entities Dev","rows":3,"metrics":["R@1","R@10","R@5"],"first_row_in_archive_order":{"model":"Fiber-B","paper":"/paper/coarse-to-fine-vision-language-pre-training","metrics":{"R@1":"87.1","R@10":"97.4","R@5":"96.1"},"code_links":[{"title":"microsoft/fiber","url":"https://github.com/microsoft/fiber"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/grounded-language-image-pre-training","title":"Grounded Language-Image Pre-training","date":"2021-12-07","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/disentangled-motif-aware-graph-learning-for","title":"Disentangled Motif-aware Graph Learning for Phrase Grounding","date":"2021-04-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-cross-modal-context-graph-for-visual","title":"Learning Cross-modal Context Graph for Visual Grounding","date":"2020-02-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/phrase-grounding-by-soft-label-chain","title":"Phrase Grounding by Soft-Label Chain Conditional Random Field","date":"2019-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/visualbert-a-simple-and-performant-baseline","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","date":"2019-08-09","rows_on_this_dataset":2,"code_links":10,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bilinear-attention-networks","title":"Bilinear Attention Networks","date":"2018-05-21","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":13,"samples_ran":4,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-diversified-and-discriminative","title":"Rethinking Diversified and Discriminative Proposal Generation for Visual Grounding","date":"2018-05-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multimodal-compact-bilinear-pooling-for","title":"Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding","date":"2016-06-06","rows_on_this_dataset":1,"code_links":10,"syntology":null},{"paper":"/paper/learning-deep-structure-preserving-image-text","title":"Learning Deep Structure-Preserving Image-Text Embeddings","date":"2015-11-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/natural-language-object-retrieval","title":"Natural Language Object Retrieval","date":"2015-11-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/grounding-of-textual-phrases-in-images-by","title":"Grounding of Textual Phrases in Images by Reconstruction","date":"2015-11-12","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/flickr30k-entities-collecting-region-to","title":"Flickr30k Entities: Collecting Region-to-Phrase Correspondences for Richer Image-to-Sentence Models","date":"2015-05-19","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-25T09:33:49+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":9,"samples_harvested":52,"samples_ran":30,"samples_unverified":22,"pointer_only_for_licence":15,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}