{"url":"/task/zero-shot-composed-image-retrieval-zs-cir","name":"Zero-Shot Composed Image Retrieval (ZS-CIR)","slug":"zero-shot-composed-image-retrieval-zs-cir","description_markdown":"Given a query composed of a reference image and a relative caption, Composed Image Retrieval (CIR) aims to retrieve target images that are visually similar to the reference one but incorporate the changes specified in the relative caption. The bi-modality of the query provides users with more precise control over the characteristics of the desired image, as some features are more easily described with language, while others can be better expressed visually. \r\n\r\n **Zero-Shot Composed Image Retrieval (ZS-CIR)** is a subtask of CIR that aims to design an approach that manages to combine the reference image and the relative caption without the need for supervised learning.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":36,"papers_with_code":25,"benchmarks":12,"benchmark_tables_in_archive":12,"benchmark_tables_shown":12,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":11,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-1","slug":"zero-shot-composed-image-retrieval-zs-cir-on-1","dataset":"CIRR","dataset_url":"/dataset/cirr","rows_in_archive":47,"metrics":["R@1","R@5","R@10","R@50","Rsubset@1"],"first_row_in_archive_order":{"model":"CoLLM (finetuned - BLIP-L/16)","paper_title":"CoLLM: A Large Language Model for Composed Image Retrieval","paper_url":"/paper/collm-a-large-language-model-for-composed","paper_date":"2025-03-25","arxiv_id":"2503.19910","code_links":[{"title":"hmchuong/CoLLM","url":"https://github.com/hmchuong/CoLLM"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on","slug":"zero-shot-composed-image-retrieval-zs-cir-on","dataset":"CIRCO","dataset_url":"/dataset/circo","rows_in_archive":43,"metrics":["mAP@10","MAP@5","mAP@50","mAP@25"],"first_row_in_archive_order":{"model":"MMRet-MLLM","paper_title":"MegaPairs: Massive Data Synthesis For Universal Multimodal Retrieval","paper_url":"/paper/megapairs-massive-data-synthesis-for","paper_date":"2024-12-19","arxiv_id":"2412.14475","code_links":[{"title":"VectorSpaceLab/MegaPairs","url":"https://github.com/VectorSpaceLab/MegaPairs"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-2","slug":"zero-shot-composed-image-retrieval-zs-cir-on-2","dataset":"Fashion IQ","dataset_url":"/dataset/fashion-iq","rows_in_archive":41,"metrics":["(Recall@10+Recall@50)/2","R@10","R@50"],"first_row_in_archive_order":{"model":"RTD + LinCIR (CLIP G/14)","paper_title":"An Efficient Post-hoc Framework for Reducing Task Discrepancy of Text Encoders for Composed Image Retrieval","paper_url":"/paper/reducing-task-discrepancy-of-text-encoders","paper_date":"2024-06-13","arxiv_id":"2406.09188","code_links":[{"title":"navervision/lincir","url":"https://github.com/navervision/lincir"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-6","slug":"zero-shot-composed-image-retrieval-zs-cir-on-6","dataset":"ImageNet-R","dataset_url":"/dataset/imagenet-r","rows_in_archive":20,"metrics":["(Recall@10+Recall@50)/2","mAP"],"first_row_in_archive_order":{"model":"MagicLens (CoCa L)","paper_title":"MagicLens: Self-Supervised Image Retrieval with Open-Ended Instructions","paper_url":"/paper/magiclens-self-supervised-image-retrieval","paper_date":"2024-03-28","arxiv_id":"2403.19651","code_links":[{"title":"google-deepmind/magiclens","url":"https://github.com/google-deepmind/magiclens"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-11","slug":"zero-shot-composed-image-retrieval-zs-cir-on-11","dataset":"GeneCIS","dataset_url":"/dataset/genecis","rows_in_archive":11,"metrics":[" A-R@1","A-R@1"],"first_row_in_archive_order":{"model":"OSrCIR (CLIP G/14)","paper_title":"Reason-before-Retrieve: One-Stage Reflective Chain-of-Thoughts for Training-Free Zero-Shot Composed Image Retrieval","paper_url":"/paper/reason-before-retrieve-one-stage-reflective","paper_date":"2024-12-15","arxiv_id":"2412.11077","code_links":[{"title":"Pter61/osrcir","url":"https://github.com/Pter61/osrcir"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-5","slug":"zero-shot-composed-image-retrieval-zs-cir-on-5","dataset":"ImageNet","dataset_url":"/dataset/imagenet","rows_in_archive":11,"metrics":["Average Recall"],"first_row_in_archive_order":{"model":"iSEARLE-XL (CLIP L/14)","paper_title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","paper_url":"/paper/isearle-improving-textual-inversion-for-zero","paper_date":"2024-05-05","arxiv_id":"2405.02951","code_links":[{"title":"miccunifi/searle","url":"https://github.com/miccunifi/searle"},{"title":"miccunifi/circo","url":"https://github.com/miccunifi/circo"}],"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-4","slug":"zero-shot-composed-image-retrieval-zs-cir-on-4","dataset":"COCO (Common Objects in Context)","dataset_url":"/dataset/coco","rows_in_archive":10,"metrics":["Actions Recall@5"],"first_row_in_archive_order":{"model":"iSEARLE-XL-OTI (CLIP L/14)","paper_title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","paper_url":"/paper/isearle-improving-textual-inversion-for-zero","paper_date":"2024-05-05","arxiv_id":"2405.02951","code_links":[{"title":"miccunifi/searle","url":"https://github.com/miccunifi/searle"},{"title":"miccunifi/circo","url":"https://github.com/miccunifi/circo"}],"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-7","slug":"zero-shot-composed-image-retrieval-zs-cir-on-7","dataset":"MiniDomainNet","dataset_url":null,"rows_in_archive":6,"metrics":["mAP"],"first_row_in_archive_order":{"model":"FreeDom (CLIP-L/14)","paper_title":"Composed Image Retrieval for Training-Free Domain Conversion","paper_url":"/paper/composed-image-retrieval-for-training-free","paper_date":"2024-12-04","arxiv_id":"2412.03297","code_links":[{"title":"nikosefth/freedom","url":"https://github.com/nikosefth/freedom"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-8","slug":"zero-shot-composed-image-retrieval-zs-cir-on-8","dataset":"NICO++","dataset_url":"/dataset/nico-1","rows_in_archive":6,"metrics":["mAP"],"first_row_in_archive_order":{"model":"FreeDom (CLIP-L/14)","paper_title":"Composed Image Retrieval for Training-Free Domain Conversion","paper_url":"/paper/composed-image-retrieval-for-training-free","paper_date":"2024-12-04","arxiv_id":"2412.03297","code_links":[{"title":"nikosefth/freedom","url":"https://github.com/nikosefth/freedom"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-9","slug":"zero-shot-composed-image-retrieval-zs-cir-on-9","dataset":"Large Time Lags Location (LTLL)","dataset_url":"/dataset/large-time-lags-location-ltll","rows_in_archive":6,"metrics":["mAP"],"first_row_in_archive_order":{"model":"FreeDom (CLIP-L/14)","paper_title":"Composed Image Retrieval for Training-Free Domain Conversion","paper_url":"/paper/composed-image-retrieval-for-training-free","paper_date":"2024-12-04","arxiv_id":"2412.03297","code_links":[{"title":"nikosefth/freedom","url":"https://github.com/nikosefth/freedom"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-10","slug":"zero-shot-composed-image-retrieval-zs-cir-on-10","dataset":"PatternCom","dataset_url":"/dataset/pattercom","rows_in_archive":2,"metrics":["mAP"],"first_row_in_archive_order":{"model":"WeiCom (RemoteCLIP)","paper_title":"Composed Image Retrieval for Remote Sensing","paper_url":"/paper/composed-image-retrieval-for-remote-sensing","paper_date":"2024-05-24","arxiv_id":"2405.15587","code_links":[{"title":"billpsomas/rscir","url":"https://github.com/billpsomas/rscir"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-3","slug":"zero-shot-composed-image-retrieval-zs-cir-on-3","dataset":"FashionIQ","dataset_url":"/dataset/fashion-iq","rows_in_archive":1,"metrics":["R@10"],"first_row_in_archive_order":{"model":"SEARLE-XL-OTI","paper_title":"Zero-Shot Composed Image Retrieval with Textual Inversion","paper_url":"/paper/zero-shot-composed-image-retrieval-with","paper_date":"2023-03-27","arxiv_id":"2303.15247","code_links":[{"title":"miccunifi/searle","url":"https://github.com/miccunifi/searle"},{"title":"miccunifi/circo","url":"https://github.com/miccunifi/circo"}],"syntology":null}}],"datasets":[{"url":"/dataset/imagenet","name":"ImageNet","full_name":"","num_papers_in_archive":15430},{"url":"/dataset/coco","name":"COCO (Common Objects in Context)","full_name":"Common Objects in Context","num_papers_in_archive":11922},{"url":"/dataset/imagenet-r","name":"ImageNet-R","full_name":"ImageNet-Rendition","num_papers_in_archive":481},{"url":"/dataset/fashion-iq","name":"Fashion IQ","full_name":"","num_papers_in_archive":102},{"url":"/dataset/cirr","name":"CIRR","full_name":"Compose Image Retrieval on Real-life images","num_papers_in_archive":61},{"url":"/dataset/circo","name":"CIRCO","full_name":"Composed Image Retrieval on Common Objects in context","num_papers_in_archive":35},{"url":"/dataset/nico-1","name":"NICO++","full_name":"","num_papers_in_archive":32},{"url":"/dataset/genecis","name":"GeneCIS","full_name":"","num_papers_in_archive":16},{"url":"/dataset/large-time-lags-location-ltll","name":"Large Time Lags Location (LTLL)","full_name":"","num_papers_in_archive":3},{"url":"/dataset/webvid-covr","name":"WebVid-CoVR","full_name":"","num_papers_in_archive":3},{"url":"/dataset/pattercom","name":"PatternCom","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/composed-image-retrieval","name":"Composed Image Retrieval (CoIR)"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":25,"of":25,"tagged_in_all":36,"items":[{"url":"/paper/semantic-editing-increment-benefits-zero-shot","title":"Semantic Editing Increment Benefits Zero-Shot Composed Image Retrieval","date":"2024-10-28","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/ldre-llm-based-divergent-reasoning-and","title":"LDRE: LLM-based Divergent Reasoning and Ensemble for Zero-Shot Composed Image Retrieval","date":"2024-07-11","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","arxiv_id":"2405.02951","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/zero-shot-composed-image-retrieval-with","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","date":"2023-03-27","arxiv_id":"2303.15247","repositories_listed":2,"syntology":null},{"url":"/paper/this-is-my-unicorn-fluffy-personalizing","title":"\"This is my unicorn, Fluffy\": Personalizing frozen vision-language representations","date":"2022-04-04","arxiv_id":"2204.01694","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/collm-a-large-language-model-for-composed","title":"CoLLM: A Large Language Model for Composed Image Retrieval","date":"2025-03-25","arxiv_id":"2503.19910","repositories_listed":1,"syntology":null},{"url":"/paper/missing-target-relevant-information","title":"Missing Target-Relevant Information Prediction with World Model for Accurate Zero-Shot Composed Image Retrieval","date":"2025-03-21","arxiv_id":"2503.17109","repositories_listed":1,"syntology":null},{"url":"/paper/imagescope-unifying-language-guided-image-1","title":"ImageScope: Unifying Language-Guided Image Retrieval via Large Multimodal Model Collective Reasoning","date":"2025-03-13","arxiv_id":"2503.10166","repositories_listed":1,"syntology":null},{"url":"/paper/megapairs-massive-data-synthesis-for","title":"MegaPairs: Massive Data Synthesis For Universal Multimodal Retrieval","date":"2024-12-19","arxiv_id":"2412.14475","repositories_listed":1,"syntology":null},{"url":"/paper/reason-before-retrieve-one-stage-reflective","title":"Reason-before-Retrieve: One-Stage Reflective Chain-of-Thoughts for Training-Free Zero-Shot Composed Image Retrieval","date":"2024-12-15","arxiv_id":"2412.11077","repositories_listed":1,"syntology":null},{"url":"/paper/composed-image-retrieval-for-training-free","title":"Composed Image Retrieval for Training-Free Domain Conversion","date":"2024-12-04","arxiv_id":"2412.03297","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-zs-cir-via-weighted-modality","title":"Training-free Zero-shot Composed Image Retrieval via Weighted Modality Fusion and Similarity","date":"2024-09-07","arxiv_id":"2409.04918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/reducing-task-discrepancy-of-text-encoders","title":"An Efficient Post-hoc Framework for Reducing Task Discrepancy of Text Encoders for Composed Image Retrieval","date":"2024-06-13","arxiv_id":"2406.09188","repositories_listed":1,"syntology":null},{"url":"/paper/composed-image-retrieval-for-remote-sensing","title":"Composed Image Retrieval for Remote Sensing","date":"2024-05-24","arxiv_id":"2405.15587","repositories_listed":1,"syntology":null},{"url":"/paper/improving-composed-image-retrieval-via","title":"Improving Composed Image Retrieval via Contrastive Learning with Scaling Positives and Negatives","date":"2024-04-17","arxiv_id":"2404.11317","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/magiclens-self-supervised-image-retrieval","title":"MagicLens: Self-Supervised Image Retrieval with Open-Ended Instructions","date":"2024-03-28","arxiv_id":"2403.19651","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/knowledge-enhanced-dual-stream-zero-shot","title":"Knowledge-Enhanced Dual-stream Zero-shot Composed Image Retrieval","date":"2024-03-24","arxiv_id":"2403.16005","repositories_listed":1,"syntology":{"n":17,"n_ran":10,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/language-only-efficient-training-of-zero-shot","title":"Language-only Efficient Training of Zero-shot Composed Image Retrieval","date":"2023-12-04","arxiv_id":"2312.01998","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":8}},{"url":"/paper/pretrain-like-you-inference-masked-tuning","title":"Pretrain like Your Inference: Masked Tuning Improves Zero-Shot Composed Image Retrieval","date":"2023-11-13","arxiv_id":"2311.07622","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/vision-by-language-for-training-free","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","date":"2023-10-13","arxiv_id":"2310.09291","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","arxiv_id":"2309.16137","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","arxiv_id":"2308.14746","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/zero-shot-composed-text-image-retrieval","title":"Zero-shot Composed Text-Image Retrieval","date":"2023-06-12","arxiv_id":"2306.07272","repositories_listed":1,"syntology":null},{"url":"/paper/compodiff-versatile-composed-image-retrieval","title":"CompoDiff: Versatile Composed Image Retrieval With Latent Diffusion","date":"2023-03-21","arxiv_id":"2303.11916","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","date":"2023-02-06","arxiv_id":"2302.03084","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}