{"url":"/task/composed-image-retrieval","name":"Composed Image Retrieval (CoIR)","slug":"composed-image-retrieval","description_markdown":"**Composed Image Retrieval (CoIR)** is the task involves retrieving images from a large database based on a query composed of multiple elements, such as text, images, and sketches. The goal is to develop algorithms that can understand and combine multiple sources of information to accurately retrieve images that match the query, extending the user’s expression ability.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":14,"papers_with_code":14,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":1,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/composed-image-retrieval-coir-on-cirr-1","slug":"composed-image-retrieval-coir-on-cirr-1","dataset":"CIRR","dataset_url":"/dataset/cirr","rows_in_archive":1,"metrics":["R@1","R@5"],"first_row_in_archive_order":{"model":"CoVR-BLIP-2","paper_title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","paper_url":"/paper/covr-learning-composed-video-retrieval-from","paper_date":"2023-08-28","arxiv_id":"2308.14746","code_links":[{"title":"lucas-ventura/CoVR","url":"https://github.com/lucas-ventura/CoVR"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/composed-image-retrieval-coir-on-fashion-iq","slug":"composed-image-retrieval-coir-on-fashion-iq","dataset":"Fashion IQ","dataset_url":"/dataset/fashion-iq","rows_in_archive":1,"metrics":["(Recall@10+Recall@50)/2","R@10","R@50"],"first_row_in_archive_order":{"model":"CoVR-BLIP-2","paper_title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","paper_url":"/paper/covr-learning-composed-video-retrieval-from","paper_date":"2023-08-28","arxiv_id":"2308.14746","code_links":[{"title":"lucas-ventura/CoVR","url":"https://github.com/lucas-ventura/CoVR"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/fashion-iq","name":"Fashion IQ","full_name":"","num_papers_in_archive":102},{"url":"/dataset/cirr","name":"CIRR","full_name":"Compose Image Retrieval on Real-life images","num_papers_in_archive":61},{"url":"/dataset/circo","name":"CIRCO","full_name":"Composed Image Retrieval on Common Objects in context","num_papers_in_archive":35},{"url":"/dataset/lasco","name":"LaSCo","full_name":"","num_papers_in_archive":7},{"url":"/dataset/webvid-covr","name":"WebVid-CoVR","full_name":"","num_papers_in_archive":3},{"url":"/dataset/pattercom","name":"PatternCom","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/zero-shot-composed-image-retrieval-zs-cir","name":"Zero-Shot Composed Image Retrieval (ZS-CIR)"}],"parent_tasks":[{"url":"/task/image-retrieval","name":"Image Retrieval"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":14,"of":14,"tagged_in_all":14,"items":[{"url":"/paper/image-retrieval-on-real-life-images-with-pre","title":"Image Retrieval on Real-life Images with Pre-trained Vision-and-Language Models","date":"2021-08-09","arxiv_id":"2108.04024","repositories_listed":3,"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","arxiv_id":"2405.02951","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/candidate-set-re-ranking-for-composed-image","title":"Candidate Set Re-ranking for Composed Image Retrieval with Dual Multi-modal Encoder","date":"2023-05-25","arxiv_id":"2305.16304","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/zero-shot-composed-image-retrieval-with","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","date":"2023-03-27","arxiv_id":"2303.15247","repositories_listed":2,"syntology":null},{"url":"/paper/conditioned-and-composed-image-retrieval","title":"Conditioned and Composed Image Retrieval Combining and Partially Fine-Tuning CLIP-Based Features","date":"2022-06-19","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/effective-conditioned-and-composed-image","title":"Effective Conditioned and Composed Image Retrieval Combining CLIP-Based Features","date":"2022-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/composed-image-retrieval-for-remote-sensing","title":"Composed Image Retrieval for Remote Sensing","date":"2024-05-24","arxiv_id":"2405.15587","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-level-prompts-benefit-composed-image","title":"Sentence-level Prompts Benefit Composed Image Retrieval","date":"2023-10-09","arxiv_id":"2310.05473","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":8}},{"url":"/paper/covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","arxiv_id":"2308.14746","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/bi-directional-training-for-composed-image","title":"Bi-directional Training for Composed Image Retrieval via Text Prompt Learning","date":"2023-03-29","arxiv_id":"2303.16604","repositories_listed":1,"syntology":null},{"url":"/paper/compodiff-versatile-composed-image-retrieval","title":"CompoDiff: Versatile Composed Image Retrieval With Latent Diffusion","date":"2023-03-21","arxiv_id":"2303.11916","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/data-roaming-and-early-fusion-for-composed","title":"Data Roaming and Quality Assessment for Composed Image Retrieval","date":"2023-03-16","arxiv_id":"2303.09429","repositories_listed":1,"syntology":null},{"url":"/paper/pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","date":"2023-02-06","arxiv_id":"2302.03084","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/composed-image-retrieval-with-text-feedback","title":"Composed Image Retrieval with Text Feedback via Multi-grained Uncertainty Regularization","date":"2022-11-14","arxiv_id":"2211.07394","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}