{"url":"/task/image-text-retrieval","name":"Image-text Retrieval","slug":"image-text-retrieval","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":248,"papers_with_code":131,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/now","name":"NoW","full_name":"Noise of Web","num_papers_in_archive":2},{"url":"/dataset/impact-patent","name":"IMPACT Patent","full_name":"A Large-scale Integrated Multimodal Patent Analysis and Creation Dataset for Design Patents","num_papers_in_archive":1},{"url":"/dataset/leafnet","name":"LeafNet","full_name":"LeafNet: A large-scale dataset for training image-text models in leaf disease identification","num_papers_in_archive":1},{"url":"/dataset/noise-of-web-now","name":"Noise of Web","full_name":"NoW","num_papers_in_archive":1},{"url":"/dataset/inpaintcoco","name":"InpaintCOCO","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":131,"tagged_in_all":248,"items":[{"url":"/paper/blip-bootstrapping-language-image-pre","title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation","date":"2022-01-28","arxiv_id":"2201.12086","repositories_listed":9,"syntology":null},{"url":"/paper/uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","arxiv_id":"1909.11740","repositories_listed":7,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/flexivit-one-model-for-all-patch-sizes","title":"FlexiViT: One Model for All Patch Sizes","date":"2022-12-15","arxiv_id":"2212.08013","repositories_listed":6,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","arxiv_id":"2107.07651","repositories_listed":6,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/scaling-up-visual-and-vision-language","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","date":"2021-02-11","arxiv_id":"2102.05918","repositories_listed":5,"syntology":{"n":10,"n_ran":8,"n_unverified":2,"n_pointer_only":9}},{"url":"/paper/mplug-effective-and-efficient-vision-language","title":"mPLUG: Effective and Efficient Vision-Language Learning by Cross-modal Skip-connections","date":"2022-05-24","arxiv_id":"2205.12005","repositories_listed":3,"syntology":null},{"url":"/paper/wit-wikipedia-based-image-text-dataset-for","title":"WIT: Wikipedia-based Image Text Dataset for Multimodal Multilingual Machine Learning","date":"2021-03-02","arxiv_id":"2103.01913","repositories_listed":3,"syntology":null},{"url":"/paper/biomedica-an-open-biomedical-image-caption","title":"BIOMEDICA: An Open Biomedical Image-Caption Archive, Dataset, and Vision-Language Models Derived from Scientific Literature","date":"2025-01-13","arxiv_id":"2501.07171","repositories_listed":2,"syntology":{"n":20,"n_ran":0,"n_unverified":20,"n_pointer_only":0}},{"url":"/paper/rwkv-clip-a-robust-vision-language","title":"RWKV-CLIP: A Robust Vision-Language Representation Learner","date":"2024-06-11","arxiv_id":"2406.06973","repositories_listed":2,"syntology":{"n":14,"n_ran":7,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/m3d-advancing-3d-medical-image-analysis-with","title":"M3D: Advancing 3D Medical Image Analysis with Multi-Modal Large Language Models","date":"2024-03-31","arxiv_id":"2404.00578","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/frozen-transformers-in-language-models-are","title":"Frozen Transformers in Language Models Are Effective Visual Encoder Layers","date":"2023-10-19","arxiv_id":"2310.12973","repositories_listed":2,"syntology":{"n":16,"n_ran":8,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/esa-external-space-attention-aggregation-for","title":"ESA: External Space Attention Aggregation for Image-Text Retrieval","date":"2023-10-10","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/contrasting-intra-modal-and-ranking-cross","title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","date":"2023-06-15","arxiv_id":"2306.08832","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","arxiv_id":"2305.11172","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/region-aware-pretraining-for-open-vocabulary","title":"Region-Aware Pretraining for Open-Vocabulary Object Detection with Vision Transformers","date":"2023-05-11","arxiv_id":"2305.07011","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/hyperbolic-image-text-representations","title":"Hyperbolic Image-Text Representations","date":"2023-04-18","arxiv_id":"2304.09172","repositories_listed":2,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":14}},{"url":"/paper/atomic-an-image-text-retrieval-test","title":"AToMiC: An Image/Text Retrieval Test Collection to Support Multimedia Content Creation","date":"2023-04-04","arxiv_id":"2304.01961","repositories_listed":2,"syntology":null},{"url":"/paper/pmc-clip-contrastive-language-image-pre","title":"PMC-CLIP: Contrastive Language-Image Pre-training using Biomedical Documents","date":"2023-03-13","arxiv_id":"2303.07240","repositories_listed":2,"syntology":{"n":13,"n_ran":6,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/uniadapter-unified-parameter-efficient","title":"UniAdapter: Unified Parameter-Efficient Transfer Learning for Cross-modal Modeling","date":"2023-02-13","arxiv_id":"2302.06605","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_unverified":2,"n_pointer_only":10}},{"url":"/paper/upop-unified-and-progressive-pruning-for","title":"UPop: Unified and Progressive Pruning for Compressing Vision-Language Transformers","date":"2023-01-31","arxiv_id":"2301.13741","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/hada-a-graph-based-amalgamation-framework-in","title":"HADA: A Graph-based Amalgamation Framework in Image-text Retrieval","date":"2023-01-11","arxiv_id":"2301.04742","repositories_listed":2,"syntology":null},{"url":"/paper/dissecting-deep-metric-learning-losses-for","title":"Dissecting Deep Metric Learning Losses for Image-Text Retrieval","date":"2022-10-21","arxiv_id":"2210.13188","repositories_listed":2,"syntology":null},{"url":"/paper/vlmo-unified-vision-language-pre-training","title":"VLMo: Unified Vision-Language Pre-Training with Mixture-of-Modality-Experts","date":"2021-11-03","arxiv_id":"2111.02358","repositories_listed":2,"syntology":null},{"url":"/paper/lightningdot-pre-training-visual-semantic","title":"LightningDOT: Pre-training Visual-Semantic Embeddings for Real-Time Image-Text Retrieval","date":"2021-03-16","arxiv_id":"2103.08784","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/gloria-a-multimodal-global-local","title":"GLoRIA: A Multimodal Global-Local Representation Learning Framework for Label-Efficient Medical Image Recognition","date":"2021-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/large-scale-adversarial-training-for-vision","title":"Large-Scale Adversarial Training for Vision-and-Language Representation Learning","date":"2020-06-11","arxiv_id":"2006.06195","repositories_listed":2,"syntology":{"n":20,"n_ran":10,"n_unverified":10,"n_pointer_only":6}},{"url":"/paper/the-neuro-symbolic-concept-learner-1","title":"The Neuro-Symbolic Concept Learner: Interpreting Scenes, Words, and Sentences From Natural Supervision","date":"2019-04-26","arxiv_id":"1904.12584","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/adding-simple-structure-at-inference-improves","title":"Adding simple structure at inference improves Vision-Language Compositionality","date":"2025-06-11","arxiv_id":"2506.09691","repositories_listed":1,"syntology":null},{"url":"/paper/flagevalmm-a-flexible-framework-for","title":"FlagEvalMM: A Flexible Framework for Comprehensive Multimodal Model Evaluation","date":"2025-06-10","arxiv_id":"2506.09081","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-attention-of-foundation-models","title":"Attacking Attention of Foundation Models Disrupts Downstream Tasks","date":"2025-06-03","arxiv_id":"2506.05394","repositories_listed":1,"syntology":null}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}