{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/conditioned-and-composed-image-retrieval","title":"Conditioned and Composed Image Retrieval Combining and Partially Fine-Tuning CLIP-Based Features","arxiv_id":null,"date":"2022-06-19","proceeding":"CVPRW 2022 6","authors":["Alberto Baldrati","Marco Bertini","Tiberio Uricchio","Alberto del Bimbo"],"abstract":"In this paper, we present an approach for conditioned and composed image retrieval based on CLIP features. In this extension of content-based image retrieval (CBIR), an image is combined with a text that provides information regarding user intentions and is relevant for application domains like e-commerce. The proposed method is based on an initial training stage where a simple combination of visual and textual features is used, to fine-tune the CLIP text encoder. Then in a second training stage, we learn a more complex combiner network that merges visual and textual features. Contrastive learning is used in both stages. The proposed approach obtains state-of-the-art performance for conditioned CBIR on the FashionIQ dataset and for composed CBIR on the more recent CIRR dataset.","url_abs":"https://openaccess.thecvf.com/content/CVPR2022W/ODRUM/html/Baldrati_Conditioned_and_Composed_Image_Retrieval_Combining_and_Partially_Fine-Tuning_CLIP-Based_CVPRW_2022_paper.html","url_pdf":"https://openaccess.thecvf.com/content/CVPR2022W/ODRUM/papers/Baldrati_Conditioned_and_Composed_Image_Retrieval_Combining_and_Partially_Fine-Tuning_CLIP-Based_CVPRW_2022_paper.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"conditioned-and-composed-image-retrieval","repo_url":"https://github.com/abaldrati/clip4cirdemo","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"conditioned-and-composed-image-retrieval","repo_url":"https://github.com/ABaldrati/CLIP4Cir","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"composed-image-retrieval","task_name":"Composed Image Retrieval (CoIR)"},{"task_slug":"content-based-image-retrieval","task_name":"Content-Based Image Retrieval"},{"task_slug":"contrastive-learning","task_name":"Contrastive Learning"},{"task_slug":"image-retrieval","task_name":"Image Retrieval"},{"task_slug":"retrieval","task_name":"Retrieval"}],"methods":[{"method_slug":"clip","method_name":"CLIP"},{"method_slug":"contrastive-learning","method_name":"Contrastive Learning"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/image-retrieval-on-cirr","task":"Image Retrieval","dataset":"CIRR","model":"CLIP4Cir (v2)","rank_in_archive_order":14,"of":17,"metrics":{"(Recall@5+Recall_subset@1)/2":"69.09"},"uses_additional_data":false},{"leaderboard":"/sota/image-retrieval-on-fashion-iq","task":"Image Retrieval","dataset":"Fashion IQ","model":"CLIP4Cir (v2)","rank_in_archive_order":14,"of":22,"metrics":{"(Recall@10+Recall@50)/2":"50.03"},"uses_additional_data":false},{"leaderboard":"/sota/image-retrieval-on-lasco","task":"Image Retrieval","dataset":"LaSCo","model":"CLIP4CIR","rank_in_archive_order":3,"of":3,"metrics":{"Recall@1 (%)":"4.01"},"uses_additional_data":false}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}