{"url":"/dataset/flickr30k","name":"Flickr30k","full_name":"Flickr30k","description_markdown":"The **Flickr30k** dataset contains 31,000 images collected from Flickr, together with 5 reference sentences provided by human annotators.\r\n\r\nSource: [Guiding Long-Short Term Memory for Image Caption Generation](https://arxiv.org/abs/1509.04942)\r\n\r\nImage Source: [Dual-Path Convolutional Image-Text Embedding with Instance Loss\r\n](https://arxiv.org/abs/1711.05535)","description_withheld":null,"homepage":"https://shannon.cs.illinois.edu/DenotationGraph/","introduced_date":"2014-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/from-image-descriptions-to-visual-denotations","title":"From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions","first_author":"Peter Young","url":null},"license":{"name":"Custom (research-only, non-commercial)","url":"https://shannon.cs.illinois.edu/DenotationGraph/#:~:text=Downloads"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Node Classification","url":"/task/node-classification","datasets_with_task":"/datasets/task/node-classification"},{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Cross-Modal Retrieval","url":"/task/cross-modal-retrieval","datasets_with_task":"/datasets/task/cross-modal-retrieval"},{"name":"Image-to-Text Retrieval","url":"/task/image-to-text-retrieval","datasets_with_task":"/datasets/task/image-to-text-retrieval"},{"name":"Phrase Grounding","url":"/task/phrase-grounding","datasets_with_task":"/datasets/task/phrase-grounding"},{"name":"Zero-shot Text-to-Image Retrieval","url":"/task/zero-shot-text-to-image-retrieval","datasets_with_task":"/datasets/task/zero-shot-text-to-image-retrieval"},{"name":"Zero-Shot Cross-Modal Retrieval","url":"/task/zero-shot-cross-modal-retrieval","datasets_with_task":"/datasets/task/zero-shot-cross-modal-retrieval"},{"name":"Semi Supervised Learning for Image Captioning","url":"/task/semi-supervised-learning-for-image-captioning","datasets_with_task":"/datasets/task/semi-supervised-learning-for-image-captioning"},{"name":"mage-to-Text Retrieval","url":"/task/mage-to-text-retrieval","datasets_with_task":"/datasets/task/mage-to-text-retrieval"},{"name":"Video Description","url":"/task/video-description","datasets_with_task":"/datasets/task/video-description"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Flickr30k","Flickr","Flickr30k Captions test","Flickr30K 1K test"],"data_loaders":[{"repo":"https://github.com/pytorch/vision","url":"https://pytorch.org/vision/stable/generated/torchvision.datasets.Flickr30k.html","frameworks":["pytorch"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#flickr30k","frameworks":["pytorch"]},{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/flickr30k-dataset","frameworks":["tf","pytorch"]}],"num_papers_in_archive":880,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/cross-modal-retrieval-on-flickr30k","task":"Cross-Modal Retrieval","dataset_variant":"Flickr30k","rows":27,"metrics":["Image-to-text R@1","Image-to-text R@5","Image-to-text R@10","Text-to-image R@1","Text-to-image R@5","Text-to-image R@10"],"first_row_in_archive_order":{"model":"X2-VLM (large)","paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","metrics":{"Image-to-text R@1":"98.8","Image-to-text R@10":"100","Image-to-text R@5":"100","Text-to-image R@1":"91.8","Text-to-image R@10":"99.5","Text-to-image R@5":"98.6"},"code_links":[{"title":"zengyan-97/x-vlm","url":"https://github.com/zengyan-97/x-vlm"},{"title":"zengyan-97/x2-vlm","url":"https://github.com/zengyan-97/x2-vlm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-cross-modal-retrieval-on-flickr30k","task":"Zero-Shot Cross-Modal Retrieval","dataset_variant":"Flickr30k","rows":22,"metrics":["Image-to-text R@1","Image-to-text R@5","Image-to-text R@10","Text-to-image R@1","Text-to-image R@5","Text-to-image R@10"],"first_row_in_archive_order":{"model":"InternVL-G","paper":"/paper/internvl-scaling-up-vision-foundation-models","metrics":{"Image-to-text R@1":"95.7","Image-to-text R@10":"99.9","Image-to-text R@5":"99.7","Text-to-image R@1":"85.0","Text-to-image R@10":"98.6","Text-to-image R@5":"97.0"},"code_links":[{"title":"opengvlab/internvl","url":"https://github.com/opengvlab/internvl"},{"title":"opengvlab/internvl-mmdetseg","url":"https://github.com/opengvlab/internvl-mmdetseg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-retrieval-on-flickr30k-1k-test","task":"Image Retrieval","dataset_variant":"Flickr30K 1K test","rows":18,"metrics":["R@1","R@5","R@10"],"first_row_in_archive_order":{"model":"X-VLM (base)","paper":"/paper/multi-grained-vision-language-pre-training","metrics":{"R@1":"86.9","R@10":"98.7","R@5":"97.3"},"code_links":[{"title":"zengyan-97/x-vlm","url":"https://github.com/zengyan-97/x-vlm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-to-text-retrieval-on-flickr30k","task":"Image-to-Text Retrieval","dataset_variant":"Flickr30k","rows":11,"metrics":["Recall@1","Recall@5","Recall@10","Recall@Sum"],"first_row_in_archive_order":{"model":"InternVL-G-FT (finetuned, w/o ranking)","paper":"/paper/internvl-scaling-up-vision-foundation-models","metrics":{"Recall@1":"97.9","Recall@10":"100","Recall@5":"100"},"code_links":[{"title":"opengvlab/internvl","url":"https://github.com/opengvlab/internvl"},{"title":"opengvlab/internvl-mmdetseg","url":"https://github.com/opengvlab/internvl-mmdetseg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-retrieval-on-flickr30k","task":"Image Retrieval","dataset_variant":"Flickr30k","rows":9,"metrics":["Recall@10","Recall@5","Recall@1","Recall@Sum","Image-to-text R@1","Image-to-text R@10","Image-to-text R@5","QPS"],"first_row_in_archive_order":{"model":"BLIP-2 ViT-G (zero-shot, 1K test set)","paper":"/paper/blip-2-bootstrapping-language-image-pre","metrics":{"Recall@1":"89.7","Recall@10":"98.9","Recall@5":"98.1"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"salesforce/lavis","url":"https://github.com/salesforce/lavis"},{"title":"thudm/visualglm-6b","url":"https://github.com/thudm/visualglm-6b"},{"title":"baaivision/eva","url":"https://github.com/baaivision/eva"},{"title":"junshutang/Make-It-3D","url":"https://github.com/junshutang/Make-It-3D"},{"title":"facebookresearch/multimodal","url":"https://github.com/facebookresearch/multimodal"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"yukw777/videoblip","url":"https://github.com/yukw777/videoblip"},{"title":"alibaba/graphtranslator","url":"https://github.com/alibaba/graphtranslator"},{"title":"gregor-ge/mblip","url":"https://github.com/gregor-ge/mblip"},{"title":"linzhiqiu/clip-flant5","url":"https://github.com/linzhiqiu/clip-flant5"},{"title":"kdr/videorag-mrr2024","url":"https://github.com/kdr/videorag-mrr2024"},{"title":"rabiulcste/vqazero","url":"https://github.com/rabiulcste/vqazero"},{"title":"jiwanchung/vlis","url":"https://github.com/jiwanchung/vlis"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-2/blip_2"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/blip_2"},{"title":"albertotestoni/ndq_visual_objects","url":"https://github.com/albertotestoni/ndq_visual_objects"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/node-classification-on-flickr","task":"Node Classification","dataset_variant":"Flickr","rows":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GCN+GAugM (Zhao et al., 2021)","paper":"/paper/data-augmentation-for-graph-neural-networks","metrics":{"Accuracy":"0.682"},"code_links":[{"title":"zhao-tong/GAug","url":"https://github.com/zhao-tong/GAug"},{"title":"andyjzhao/wsdm23-gsr","url":"https://github.com/andyjzhao/wsdm23-gsr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-flickr30k-captions-test","task":"Image Captioning","dataset_variant":"Flickr30k Captions test","rows":7,"metrics":["BLEU-4","CIDEr","METEOR","SPICE"],"first_row_in_archive_order":{"model":"Unified VLP","paper":"/paper/unified-vision-language-pre-training-for","metrics":{"BLEU-4":"30.1","CIDEr":"67.4","METEOR":"23","SPICE":"17"},"code_links":[{"title":"rmokady/clip_prefix_caption","url":"https://github.com/rmokady/clip_prefix_caption"},{"title":"LuoweiZhou/VLP","url":"https://github.com/LuoweiZhou/VLP"},{"title":"WebQnA/WebQA_Baseline","url":"https://github.com/WebQnA/WebQA_Baseline"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/phrase-grounding-on-flickr30k","task":"Phrase Grounding","dataset_variant":"Flickr30k","rows":3,"metrics":["Pointing Game Accuracy"],"first_row_in_archive_order":{"model":"GBS Ensemble + 12-in-1","paper":"/paper/detector-free-weakly-supervised-grounding-by","metrics":{"Pointing Game Accuracy":"85.9"},"code_links":[{"title":"aarbelle/GroundingBySeparation","url":"https://github.com/aarbelle/GroundingBySeparation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semi-supervised-learning-for-image-captioning-2","task":"Semi Supervised Learning for Image Captioning","dataset_variant":"Flickr30k","rows":1,"metrics":["CIDEr"],"first_row_in_archive_order":{"model":"CapDec","paper":"/paper/text-only-training-for-image-captioning-using","metrics":{"CIDEr":"39.1"},"code_links":[{"title":"davidhuji/capdec","url":"https://github.com/davidhuji/capdec"},{"title":"zelaki/wsac","url":"https://github.com/zelaki/wsac"},{"title":"avitej-iyer/CapDec-Recreation","url":"https://github.com/avitej-iyer/CapDec-Recreation"},{"title":"uriberger/re_cap","url":"https://github.com/uriberger/re_cap"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cosmos-cross-modality-self-distillation-for","title":"COSMOS: Cross-Modality Self-Distillation for Vision Language Pre-training","date":"2024-12-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/3shnet-boosting-image-sentence-retrieval-via","title":"3SHNet: Boosting Image-Sentence Retrieval via Visual Semantic-Spatial Self-Highlighting","date":"2024-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":10,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dynamic-self-adaptive-multiscale-distillation","title":"Dynamic Self-adaptive Multiscale Distillation from Pre-trained Multimodal Large Model for Efficient Cross-modal Representation Learning","date":"2024-04-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/boldsymbol-m-2-encoder-advancing-bilingual","title":"M2-Encoder: Advancing Bilingual Image-Text Understanding by Large-scale Efficient Pretraining","date":"2024-01-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/region-aware-pretraining-for-open-vocabulary","title":"Region-Aware Pretraining for Open-Vocabulary Object Detection with Vision Transformers","date":"2023-05-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/plug-and-play-regulators-for-image-text","title":"Plug-and-Play Regulators for Image-Text Matching","date":"2023-03-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","rows_on_this_dataset":4,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hada-a-graph-based-amalgamation-framework-in","title":"HADA: A Graph-based Amalgamation Framework in Image-text Retrieval","date":"2023-01-11","rows_on_this_dataset":5,"code_links":2,"syntology":null},{"paper":"/paper/napreg-nouns-as-proxies-regularization-for","title":"NAPReg: Nouns As Proxies Regularization for Semantically Aware Cross-Modal Embeddings","date":"2023-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reproducible-scaling-laws-for-contrastive","title":"Reproducible scaling laws for contrastive language-image learning","date":"2022-12-14","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/altclip-altering-the-language-encoder-in-clip","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","date":"2022-11-12","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-only-training-for-image-captioning-using","title":"Text-Only Training for Image Captioning using Noise-Injected CLIP","date":"2022-11-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dissecting-deep-metric-learning-losses-for","title":"Dissecting Deep Metric Learning Losses for Image-Text Retrieval","date":"2022-10-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-comprehensive-study-on-large-scale-graph","title":"A Comprehensive Study on Large-Scale Graph Training: Benchmarking and Rethinking","date":"2022-10-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ernie-vil-2-0-multi-view-contrastive-learning","title":"ERNIE-ViL 2.0: Multi-view Contrastive Learning for Image-Text Pre-training","date":"2022-09-30","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/image-as-a-foreign-language-beit-pretraining","title":"Image as a Foreign Language: BEiT Pretraining for All Vision and Vision-Language Tasks","date":"2022-08-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/language-models-are-general-purpose","title":"Language Models are General-Purpose Interfaces","date":"2022-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":18,"samples_unverified":6,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vista-vision-and-scene-text-aggregation-for","title":"ViSTA: Vision and Scene Text Aggregation for Cross-Modal Retrieval","date":"2022-03-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/florence-a-new-foundation-model-for-computer","title":"Florence: A New Foundation Model for Computer Vision","date":"2021-11-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-good-prompt-is-worth-millions-of-parameters","title":"A Good Prompt Is Worth Millions of Parameters: Low-resource Prompt-based Learning for Vision-Language Models","date":"2021-10-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-deep-local-and-global-scene-graph-matching","title":"A Deep Local and Global Scene-Graph Matching for Image-Text Retrieval","date":"2021-06-04","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/learning-relation-alignment-for-calibrated","title":"Learning Relation Alignment for Calibrated Cross-modal Retrieval","date":"2021-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detector-free-weakly-supervised-grounding-by","title":"Detector-Free Weakly Supervised Grounding by Separation","date":"2021-04-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","rows_on_this_dataset":1,"code_links":82,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-up-visual-and-vision-language","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","date":"2021-02-11","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifying-vision-and-language-tasks-via-text","title":"Unifying Vision-and-Language Tasks via Text Generation","date":"2021-02-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/similarity-reasoning-and-filtration-for-image","title":"Similarity Reasoning and Filtration for Image-Text Matching","date":"2021-01-05","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visualsparta-sparse-transformer-fragment","title":"VisualSparta: An Embarrassingly Simple Approach to Large-scale Text-to-Image Search with Weighted Bag-of-words","date":"2021-01-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/fine-grained-visual-textual-alignment-for","title":"Fine-grained Visual Textual Alignment for Cross-Modal Retrieval using Transformer Encoders","date":"2020-08-12","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":2,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/data-augmentation-for-graph-neural-networks","title":"Data Augmentation for Graph Neural Networks","date":"2020-06-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/graph-structured-network-for-image-text","title":"Graph Structured Network for Image-Text Matching","date":"2020-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/imram-iterative-matching-with-recurrent","title":"IMRAM: Iterative Matching with Recurrent Attention Memory for Cross-Modal Image-Text Retrieval","date":"2020-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/imagebert-cross-modal-pre-training-with-large","title":"ImageBERT: Cross-modal Pre-training with Large-scale Weak-supervised Image-Text Data","date":"2020-01-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unified-vision-language-pre-training-for","title":"Unified Vision-Language Pre-Training for Image Captioning and VQA","date":"2019-09-24","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/camp-cross-modal-adaptive-message-passing-for","title":"CAMP: Cross-Modal Adaptive Message Passing for Text-Image Retrieval","date":"2019-09-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visual-semantic-reasoning-for-image-text","title":"Visual Semantic Reasoning for Image-Text Matching","date":"2019-09-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/demo-net-degree-specific-graph-neural","title":"DEMO-Net: Degree-specific Graph Neural Networks for Node and Graph Classification","date":"2019-06-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multi-level-multimodal-common-semantic-space","title":"Multi-level Multimodal Common Semantic Space for Image-Phrase Grounding","date":"2018-11-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-cross-modal-projection-learning-for","title":"Deep Cross-Modal Projection Learning for Image-Text Matching","date":"2018-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stacked-cross-attention-for-image-text","title":"Stacked Cross Attention for Image-Text Matching","date":"2018-03-21","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":7,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deeper-insights-into-graph-convolutional","title":"Deeper Insights into Graph Convolutional Networks for Semi-Supervised Learning","date":"2018-01-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-semantic-concepts-and-order-for","title":"Learning Semantic Concepts and Order for Image and Sentence Matching","date":"2017-12-06","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/dual-path-convolutional-image-text-embedding","title":"Dual-Path Convolutional Image-Text Embeddings with Instance Loss","date":"2017-11-15","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/graph-attention-networks","title":"Graph Attention Networks","date":"2017-10-30","rows_on_this_dataset":1,"code_links":93,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":106,"samples_ran":50,"samples_unverified":56,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vse-improving-visual-semantic-embeddings-with","title":"VSE++: Improving Visual-Semantic Embeddings with Hard Negatives","date":"2017-07-18","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/paying-more-attention-to-saliency-image","title":"Paying More Attention to Saliency: Image Captioning with Saliency and Context Attention","date":"2017-06-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/inductive-representation-learning-on-large","title":"Inductive Representation Learning on Large Graphs","date":"2017-06-07","rows_on_this_dataset":1,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/instance-aware-image-and-sentence-matching","title":"Instance-aware Image and Sentence Matching with Selective Multimodal LSTM","date":"2016-11-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dual-attention-networks-for-multimodal","title":"Dual Attention Networks for Multimodal Reasoning and Matching","date":"2016-11-02","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/semi-supervised-classification-with-graph","title":"Semi-Supervised Classification with Graph Convolutional Networks","date":"2016-09-09","rows_on_this_dataset":2,"code_links":55,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":31,"samples_unverified":27,"pointer_only_for_licence":22,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/linking-image-and-text-with-2-way-nets","title":"Linking Image and Text with 2-Way Nets","date":"2016-08-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-deep-structure-preserving-image-text","title":"Learning Deep Structure-Preserving Image-Text Embeddings","date":"2015-11-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/flickr30k-entities-collecting-region-to","title":"Flickr30k Entities: Collecting Region-to-Phrase Correspondences for Richer Image-to-Sentence Models","date":"2015-05-19","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-convolutional-neural-networks-for","title":"Multimodal Convolutional Neural Networks for Matching Image and Sentence","date":"2015-04-23","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/deep-visual-semantic-alignments-for","title":"Deep Visual-Semantic Alignments for Generating Image Descriptions","date":"2014-12-07","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":37,"samples_harvested":463,"samples_ran":238,"samples_unverified":225,"pointer_only_for_licence":157,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}