{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/image-text-retrieval/papers/3","list_of":"/task/image-text-retrieval","task":"Image-text Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,248],"of":248,"counts":{"archive_papers_tagged":248,"with_a_code_link":131,"where_syntology_ran_a_sample":49,"not_listed_spam_title":0,"listed":248,"listed_where_code_ran":49,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":44,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":44,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/image-text-retrieval","prev":"/task/image-text-retrieval/papers/2","next":null,"papers":[{"url":null,"slug":"vl-match-enhancing-vision-language","title":"VL-Match: Enhancing Vision-Language Pretraining with Token-Level and Instance-Level Matching","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-image-captioning-for-edge-devices","title":"Efficient Image Captioning for Edge Devices","date":"2022-12-18","arxiv_id":"2212.08985","repositories_listed":0,"syntology":null},{"url":null,"slug":"hgan-hierarchical-graph-alignment-network-for","title":"HGAN: Hierarchical Graph Alignment Network for Image-Text Retrieval","date":"2022-12-16","arxiv_id":"2212.08281","repositories_listed":0,"syntology":null},{"url":null,"slug":"nlip-noise-robust-language-image-pre-training","title":"NLIP: Noise-robust Language-Image Pre-training","date":"2022-12-14","arxiv_id":"2212.07086","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-semantic-joint-decoupling-network-for","title":"Scale-Semantic Joint Decoupling Network for Image-text Retrieval in Remote Sensing","date":"2022-12-12","arxiv_id":"2212.05752","repositories_listed":0,"syntology":null},{"url":"/paper/masked-contrastive-pre-training-for-efficient","slug":"masked-contrastive-pre-training-for-efficient","title":"Masked Contrastive Pre-Training for Efficient Video-Text Retrieval","date":"2022-12-02","arxiv_id":"2212.00986","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-negative-text-replay-for-continual","title":"Generative Negative Text Replay for Continual Vision-Language Pretraining","date":"2022-10-31","arxiv_id":"2210.17322","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-text-retrieval-with-binary-and","title":"Image-Text Retrieval with Binary and Continuous Label Supervision","date":"2022-10-20","arxiv_id":"2210.11319","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpl-counterfactual-prompt-learning-for-vision","title":"CPL: Counterfactual Prompt Learning for Vision and Language Models","date":"2022-10-19","arxiv_id":"2210.10362","repositories_listed":0,"syntology":null},{"url":null,"slug":"mamo-masked-multimodal-modeling-for-fine","title":"MAMO: Masked Multimodal Modeling for Fine-Grained Vision-Language Representation Learning","date":"2022-10-09","arxiv_id":"2210.04183","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-embed-semantic-similarity-for","title":"Learning to embed semantic similarity for joint image-text retrieval","date":"2022-10-07","arxiv_id":"2210.03838","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-multilingual-multi-modal-pre","title":"Efficient Multilingual Multi-modal Pre-training through Triple Contrastive Loss","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/re-imagen-retrieval-augmented-text-to-image","slug":"re-imagen-retrieval-augmented-text-to-image","title":"Re-Imagen: Retrieval-Augmented Text-to-Image Generator","date":"2022-09-29","arxiv_id":"2209.14491","repositories_listed":0,"syntology":null},{"url":null,"slug":"revising-image-text-retrieval-via-multi-modal","title":"Revising Image-Text Retrieval via Multi-Modal Entailment","date":"2022-08-22","arxiv_id":"2208.10126","repositories_listed":0,"syntology":null},{"url":null,"slug":"coder-coupled-diversity-sensitive-momentum","title":"CODER: Coupled Diversity-Sensitive Momentum Contrastive Learning for Image-Text Retrieval","date":"2022-08-21","arxiv_id":"2208.09843","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlmae-vision-language-masked-autoencoder","title":"VLMAE: Vision-Language Masked Autoencoder","date":"2022-08-19","arxiv_id":"2208.09374","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-contrastive-distillation-for-image","title":"Dynamic Contrastive Distillation for Image-Text Retrieval","date":"2022-07-04","arxiv_id":"2207.01426","repositories_listed":0,"syntology":null},{"url":null,"slug":"vl-beit-generative-vision-language","title":"VL-BEiT: Generative Vision-Language Pretraining","date":"2022-06-02","arxiv_id":"2206.01127","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-based-learning-for-unpaired-image","title":"Prompt-based Learning for Unpaired Image Captioning","date":"2022-05-26","arxiv_id":"2205.13125","repositories_listed":0,"syntology":null},{"url":"/paper/crossmodal-3600-a-massively-multilingual-1","slug":"crossmodal-3600-a-massively-multilingual-1","title":"Crossmodal-3600: A Massively Multilingual Multimodal Evaluation Dataset","date":"2022-05-25","arxiv_id":"2205.12522","repositories_listed":0,"syntology":null},{"url":null,"slug":"hivlp-hierarchical-vision-language-pre","title":"HiVLP: Hierarchical Vision-Language Pre-Training for Fast Image-Text Retrieval","date":"2022-05-24","arxiv_id":"2205.12105","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-learning-for-image-retrieval-with","title":"Progressive Learning for Image Retrieval with Hybrid-Modality Queries","date":"2022-04-24","arxiv_id":"2204.11212","repositories_listed":0,"syntology":null},{"url":"/paper/cots-collaborative-two-stream-vision-language","slug":"cots-collaborative-two-stream-vision-language","title":"COTS: Collaborative Two-Stream Vision-Language Pre-Training Model for Cross-Modal Retrieval","date":"2022-04-15","arxiv_id":"2204.07441","repositories_listed":0,"syntology":null},{"url":"/paper/robust-cross-modal-representation-learning","slug":"robust-cross-modal-representation-learning","title":"Robust Cross-Modal Representation Learning with Progressive Self-Distillation","date":"2022-04-10","arxiv_id":"2204.04588","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-text-retrieval-a-survey-on-recent","title":"Image-text Retrieval: A Survey on Recent Research and Development","date":"2022-03-28","arxiv_id":"2203.14713","repositories_listed":0,"syntology":null},{"url":null,"slug":"loopitr-combining-dual-and-cross-encoder","title":"LoopITR: Combining Dual and Cross Encoder Architectures for Image-Text Retrieval","date":"2022-03-10","arxiv_id":"2203.05465","repositories_listed":0,"syntology":null},{"url":null,"slug":"commercemm-large-scale-commerce-multimodal","title":"CommerceMM: Large-Scale Commerce MultiModal Representation Learning with Omni Retrieval","date":"2022-02-15","arxiv_id":"2202.07247","repositories_listed":0,"syntology":null},{"url":null,"slug":"negative-sample-is-negative-in-its-own-way-1","title":"Negative Sample is Negative in Its Own Way: Tailoring Negative Sentences for Image-Text Retrieval","date":"2021-12-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-multimodal-pre-training-and-prompt","title":"Unified Multimodal Pre-training and Prompt-based Tuning for Vision-Language Understanding and Generation","date":"2021-12-10","arxiv_id":"2112.05587","repositories_listed":0,"syntology":null},{"url":null,"slug":"ufo-a-unified-transformer-for-vision-language","title":"UFO: A UniFied TransfOrmer for Vision-Language Representation Learning","date":"2021-11-19","arxiv_id":"2111.10023","repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-phrase-level-semantic-labels-to-1","title":"Constructing Phrase-level Semantic Labels to Form Multi-GrainedSupervision for Image-Text Retrieval","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"swamp-swapped-assignment-of-multi-modal-pairs","title":"SwAMP: Swapped Assignment of Multi-Modal Pairs for Cross-Modal Retrieval","date":"2021-11-10","arxiv_id":"2111.05814","repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-phrase-level-semantic-labels-to","title":"Constructing Phrase-level Semantic Labels to Form Multi-Grained Supervision for Image-Text Retrieval","date":"2021-09-12","arxiv_id":"2109.05523","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-inter-modality-visual-parsing-with","title":"Probing Inter-modality: Visual Parsing with Self-Attention for Vision-Language Pre-training","date":"2021-06-25","arxiv_id":"2106.13488","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-of-visual-semantic-embedding-methods","title":"Survey of Visual-Semantic Embedding Methods for Zero-Shot Image Retrieval","date":"2021-05-16","arxiv_id":"2105.07391","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-lottery-tickets-with-vision-and","title":"Playing Lottery Tickets with Vision and Language","date":"2021-04-23","arxiv_id":"2104.11832","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-in-cross-modal-retrieval","title":"Continual learning in cross-modal retrieval","date":"2021-04-14","arxiv_id":"2104.06806","repositories_listed":0,"syntology":null},{"url":null,"slug":"uc2-universal-cross-lingual-cross-modal","title":"UC2: Universal Cross-lingual Cross-modal Vision-and-Language Pre-training","date":"2021-04-01","arxiv_id":"2104.00332","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-multi-modal-nonlinear-embeddings","title":"Learning Multi-Modal Nonlinear Embeddings: Performance Bounds and an Algorithm","date":"2020-06-03","arxiv_id":"2006.02330","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-attention-network-for-image","title":"Context-Aware Attention Network for Image-Text Retrieval","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/xgpt-cross-modal-generative-pre-training-for","slug":"xgpt-cross-modal-generative-pre-training-for","title":"XGPT: Cross-modal Generative Pre-Training for Image Captioning","date":"2020-03-03","arxiv_id":"2003.01473","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniter-learning-universal-image-text","title":"UNITER: Learning UNiversal Image-TExt Representations","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/unicoder-vl-a-universal-encoder-for-vision","slug":"unicoder-vl-a-universal-encoder-for-vision","title":"Unicoder-VL: A Universal Encoder for Vision and Language by Cross-modal Pre-training","date":"2019-08-16","arxiv_id":"1908.06066","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-semantic-multimodal-hashing-network-for","title":"Deep Semantic Multimodal Hashing Network for Scalable Image-Text and Video-Text Retrievals","date":"2019-01-09","arxiv_id":"1901.02662","repositories_listed":0,"syntology":null},{"url":null,"slug":"webly-supervised-joint-embedding-for-cross-1","title":"Webly Supervised Joint Embedding for Cross-Modal lmage-Text Retrieval","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"webly-supervised-joint-embedding-for-cross","title":"Webly Supervised Joint Embedding for Cross-Modal Image-Text Retrieval","date":"2018-08-23","arxiv_id":"1808.07793","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-imagine-and-match-improving-textual","title":"Look, Imagine and Match: Improving Textual-Visual Cross-Modal Retrieval with Generative Models","date":"2017-11-17","arxiv_id":"1711.06420","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymmetrically-weighted-cca-and-hierarchical","title":"Asymmetrically Weighted CCA And Hierarchical Kernel Sentence Embedding For Image & Text Retrieval","date":"2015-11-19","arxiv_id":"1511.06267","repositories_listed":0,"syntology":null}],"record_sha256":"16a28ea3651a3d32f0b5b8cae8428c019f66cb365b25b5846dce2d0b751b9221","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}