{"url":"/dataset/circo","name":"CIRCO","full_name":"Composed Image Retrieval on Common Objects in context","description_markdown":"**CIRCO (Composed Image Retrieval on Common Objects in context)** is an open-domain benchmarking dataset for Composed Image Retrieval (CIR) based on real-world images from COCO 2017 unlabeled set. It is the first CIR dataset with multiple ground truths and aims to address the problem of false negatives in existing datasets. CIRCO comprises a total of 1020 queries, randomly divided into 220 and 800 for the validation and test set, respectively, with an average of 4.53 ground truths per query.\r\n\r\nSource: [Zero-Shot Composed Image Retrieval with Textual Inversion](https://arxiv.org/pdf/2303.15247v1.pdf)\r\n\r\nImage Source: [Zero-Shot Composed Image Retrieval with Textual Inversion](https://arxiv.org/pdf/2303.15247v1.pdf)","description_withheld":null,"homepage":"https://github.com/miccunifi/CIRCO","introduced_date":"2023-03-27","introduced_date_note":null,"introduced_by":{"paper":"/paper/zero-shot-composed-image-retrieval-with","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","first_author":"Alberto Baldrati","url":null},"license":{"name":"Creative Commons BY-NC 4.0","url":"https://github.com/miccunifi/CIRCO/blob/main/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Composed Image Retrieval (CoIR)","url":"/task/composed-image-retrieval","datasets_with_task":"/datasets/task/composed-image-retrieval"},{"name":"Zero-Shot Composed Image Retrieval (ZS-CIR)","url":"/task/zero-shot-composed-image-retrieval-zs-cir","datasets_with_task":"/datasets/task/zero-shot-composed-image-retrieval-zs-cir"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["CIRCO"],"data_loaders":[{"repo":"https://github.com/miccunifi/circo","url":"https://github.com/miccunifi/CIRCO/blob/main/README.md","frameworks":["pytorch"]}],"num_papers_in_archive":35,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset_variant":"CIRCO","rows":43,"metrics":["mAP@10","MAP@5","mAP@50","mAP@25"],"first_row_in_archive_order":{"model":"MMRet-MLLM","paper":"/paper/megapairs-massive-data-synthesis-for","metrics":{"mAP@10":"43.4"},"code_links":[{"title":"VectorSpaceLab/MegaPairs","url":"https://github.com/VectorSpaceLab/MegaPairs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/collm-a-large-language-model-for-composed","title":"CoLLM: A Large Language Model for Composed Image Retrieval","date":"2025-03-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/imagescope-unifying-language-guided-image-1","title":"ImageScope: Unifying Language-Guided Image Retrieval via Large Multimodal Model Collective Reasoning","date":"2025-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scot-self-supervised-contrastive-pretraining","title":"SCOT: Self-Supervised Contrastive Pretraining For Zero-Shot Compositional Retrieval","date":"2025-01-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/megapairs-massive-data-synthesis-for","title":"MegaPairs: Massive Data Synthesis For Universal Multimodal Retrieval","date":"2024-12-19","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/reason-before-retrieve-one-stage-reflective","title":"Reason-before-Retrieve: One-Stage Reflective Chain-of-Thoughts for Training-Free Zero-Shot Composed Image Retrieval","date":"2024-12-15","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/imagine-and-seek-improving-composed-image","title":"Imagine and Seek: Improving Composed Image Retrieval with an Imagined Proxy","date":"2024-11-24","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/semantic-editing-increment-benefits-zero-shot","title":"Semantic Editing Increment Benefits Zero-Shot Composed Image Retrieval","date":"2024-10-28","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/ldre-llm-based-divergent-reasoning-and","title":"LDRE: LLM-based Divergent Reasoning and Ensemble for Zero-Shot Composed Image Retrieval","date":"2024-07-11","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/reducing-task-discrepancy-of-text-encoders","title":"An Efficient Post-hoc Framework for Reducing Task Discrepancy of Text Encoders for Composed Image Retrieval","date":"2024-06-13","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/magiclens-self-supervised-image-retrieval","title":"MagicLens: Self-Supervised Image Retrieval with Open-Ended Instructions","date":"2024-03-28","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-only-efficient-training-of-zero-shot","title":"Language-only Efficient Training of Zero-shot Composed Image Retrieval","date":"2023-12-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pretrain-like-you-inference-masked-tuning","title":"Pretrain like Your Inference: Masked Tuning Improves Zero-Shot Composed Image Retrieval","date":"2023-11-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-by-language-for-training-free","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","date":"2023-10-13","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-composed-image-retrieval-with","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","date":"2023-03-27","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/compodiff-versatile-composed-image-retrieval","title":"CompoDiff: Versatile Composed Image Retrieval With Latent Diffusion","date":"2023-03-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","date":"2023-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/this-is-my-unicorn-fluffy-personalizing","title":"\"This is my unicorn, Fluffy\": Personalizing frozen vision-language representations","date":"2022-04-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":35,"samples_ran":21,"samples_unverified":14,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}