{"url":"/task/zero-shot-text-to-image-retrieval","name":"Zero-shot Text-to-Image Retrieval","slug":"zero-shot-text-to-image-retrieval","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":15,"papers_with_code":11,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":7,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/coco","name":"COCO (Common Objects in Context)","full_name":"Common Objects in Context","num_papers_in_archive":11922},{"url":"/dataset/flickr30k","name":"Flickr30k","full_name":"Flickr30k","num_papers_in_archive":880},{"url":"/dataset/xm-3600","name":"XM 3600","full_name":"Crossmodal 3600","num_papers_in_archive":28},{"url":"/dataset/coco-cn","name":"COCO-CN","full_name":"","num_papers_in_archive":21},{"url":"/dataset/fewsol","name":"FewSOL","full_name":"A Dataset for Few-Shot Object Learning in Robotic Environments","num_papers_in_archive":4},{"url":"/dataset/ilias","name":"ILIAS","full_name":"ILIAS: Instance-Level Image retrieval At Scale","num_papers_in_archive":1},{"url":"/dataset/coco-facet","name":"COCO-Facet","full_name":"COCO-Facet","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":11,"of":11,"tagged_in_all":15,"items":[{"url":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","arxiv_id":"2103.00020","repositories_listed":82,"syntology":{"n":20,"n_ran":16,"n_unverified":4,"n_pointer_only":16}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/sigmoid-loss-for-language-image-pre-training","title":"Sigmoid Loss for Language Image Pre-Training","date":"2023-03-27","arxiv_id":"2303.15343","repositories_listed":11,"syntology":{"n":29,"n_ran":12,"n_unverified":17,"n_pointer_only":25}},{"url":"/paper/flava-a-foundational-language-and-vision","title":"FLAVA: A Foundational Language And Vision Alignment Model","date":"2021-12-08","arxiv_id":"2112.04482","repositories_listed":4,"syntology":null},{"url":"/paper/one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","arxiv_id":"2305.11172","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/magiclens-self-supervised-image-retrieval","title":"MagicLens: Self-Supervised Image Retrieval with Open-Ended Instructions","date":"2024-03-28","arxiv_id":"2403.19651","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/boldsymbol-m-2-encoder-advancing-bilingual","title":"M2-Encoder: Advancing Bilingual Image-Text Understanding by Large-scale Efficient Pretraining","date":"2024-01-29","arxiv_id":"2401.15896","repositories_listed":1,"syntology":null},{"url":"/paper/linguistic-aware-patch-slimming-framework-for","title":"Linguistic-Aware Patch Slimming Framework for Fine-grained Cross-Modal Alignment","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/chinese-clip-contrastive-vision-language","title":"Chinese CLIP: Contrastive Vision-Language Pretraining in Chinese","date":"2022-11-02","arxiv_id":"2211.01335","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/ernie-vil-2-0-multi-view-contrastive-learning","title":"ERNIE-ViL 2.0: Multi-view Contrastive Learning for Image-Text Pre-training","date":"2022-09-30","arxiv_id":"2209.15270","repositories_listed":1,"syntology":null},{"url":"/paper/zscrgan-a-gan-based-expectation-maximization","title":"ZSCRGAN: A GAN-based Expectation Maximization Model for Zero-Shot Retrieval of Images from Textual Descriptions","date":"2020-07-23","arxiv_id":"2007.12212","repositories_listed":1,"syntology":null}],"syntology_records":6,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}