{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","arxiv_id":"2302.03084","date":"2023-02-06","proceeding":"CVPR 2023 1","authors":["Kuniaki Saito","Kihyuk Sohn","Xiang Zhang","Chun-Liang Li","Chen-Yu Lee","Kate Saenko","Tomas Pfister"],"abstract":"In Composed Image Retrieval (CIR), a user combines a query image with text to describe their intended target. Existing methods rely on supervised learning of CIR models using labeled triplets consisting of the query image, text specification, and the target image. Labeling such triplets is expensive and hinders broad applicability of CIR. In this work, we propose to study an important task, Zero-Shot Composed Image Retrieval (ZS-CIR), whose goal is to build a CIR model without requiring labeled triplets for training. To this end, we propose a novel method, called Pic2Word, that requires only weakly labeled image-caption pairs and unlabeled image datasets to train. Unlike existing supervised CIR models, our model trained on weakly labeled or unlabeled datasets shows strong generalization across diverse ZS-CIR tasks, e.g., attribute editing, object composition, and domain conversion. Our approach outperforms several supervised CIR methods on the common CIR benchmark, CIRR and Fashion-IQ. Code will be made publicly available at https://github.com/google-research/composed_image_retrieval.","url_abs":"https://arxiv.org/abs/2302.03084v2","url_pdf":"https://arxiv.org/pdf/2302.03084v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"pic2word-mapping-pictures-to-words-for-zero","repo_url":"https://github.com/google-research/composed_image_retrieval","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"attribute","task_name":"Attribute"},{"task_slug":"composed-image-retrieval","task_name":"Composed Image Retrieval (CoIR)"},{"task_slug":"image-retrieval","task_name":"Image Retrieval"},{"task_slug":"retrieval","task_name":"Retrieval"},{"task_slug":"zero-shot-composed-image-retrieval-zs-cir","task_name":"Zero-Shot Composed Image Retrieval (ZS-CIR)"},{"task_slug":"zero-shot-image-retrieval","task_name":"Zero-shot Image Retrieval"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset":"CIRCO","model":"Pic2Word","rank_in_archive_order":41,"of":43,"metrics":{"mAP@10":"9.51"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-1","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset":"CIRR","model":"Pic2Word","rank_in_archive_order":46,"of":47,"metrics":{"R@5":"51.70"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-4","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset":"COCO (Common Objects in Context)","model":"Pic2Word","rank_in_archive_order":9,"of":10,"metrics":{"Actions Recall@5":"24.8"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-2","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset":"Fashion IQ","model":"Pic2Word","rank_in_archive_order":37,"of":41,"metrics":{"(Recall@10+Recall@50)/2":"34.20"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-5","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset":"ImageNet","model":"Pic2Word","rank_in_archive_order":7,"of":11,"metrics":{"Average Recall":"18.85"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-6","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset":"ImageNet-R","model":"Pic2Word","rank_in_archive_order":10,"of":20,"metrics":{"(Recall@10+Recall@50)/2":"16.65"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2302.03084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03084"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/google-research/composed_image_retrieval","reach":null},{"provenance":"deterministic:regex_extraction","url":"https://github.com/googleresearch/composed_image_retrieval","reach":{"status":"gone","observed_at":"2026-09-17","how":"tree_404+repo_404"}}],"summary":{"ran":1},"by_repo_kind":{"official":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"87023f1da7e3a4b9","entry":"IM2TEXT","repo":"google-research/composed_image_retrieval","repo_kind":"official","path":"model/model.py","file_url":"https://github.com/google-research/composed_image_retrieval/blob/HEAD/model/model.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"87023f1da7e3a4b9"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}