{"url":"/dataset/fashion-iq","name":"Fashion IQ","full_name":null,"description_markdown":"Fashion IQ support and advance research on interactive fashion image retrieval. Fashion IQ is the first fashion dataset to provide human-generated captions that distinguish similar pairs of garment images together with side-information consisting of real-world product descriptions and derived visual attribute labels for these images.\r\n\r\nSource: [Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback](/paper/the-fashion-iq-dataset-retrieving-images-by)","description_withheld":null,"homepage":"https://github.com/XiaoxiaoGuo/fashion-iq","introduced_date":"2019-05-30","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-fashion-iq-dataset-retrieving-images-by","title":"Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback","first_author":"Hui Wu","url":null},"license":null,"modalities":[],"tasks":[{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Image Retrieval with Multi-Modal Query","url":"/task/multi-modal","datasets_with_task":"/datasets/task/multi-modal"},{"name":"Composed Image Retrieval (CoIR)","url":"/task/composed-image-retrieval","datasets_with_task":"/datasets/task/composed-image-retrieval"},{"name":"Zero-Shot Composed Image Retrieval (ZS-CIR)","url":"/task/zero-shot-composed-image-retrieval-zs-cir","datasets_with_task":"/datasets/task/zero-shot-composed-image-retrieval-zs-cir"},{"name":"Virtual Try-on","url":"/task/virtual-try-on","datasets_with_task":"/datasets/task/virtual-try-on"}],"languages":[],"variants":["FashionIQ","Fashion IQ"],"data_loaders":[{"repo":"https://github.com/XiaoxiaoGuo/fashion-iq","url":"https://github.com/XiaoxiaoGuo/fashion-iq","frameworks":["pytorch"]}],"num_papers_in_archive":102,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-2","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset_variant":"Fashion IQ","rows":41,"metrics":["(Recall@10+Recall@50)/2","R@10","R@50"],"first_row_in_archive_order":{"model":"RTD + LinCIR (CLIP G/14)","paper":"/paper/reducing-task-discrepancy-of-text-encoders","metrics":{"(Recall@10+Recall@50)/2":"56.74"},"code_links":[{"title":"navervision/lincir","url":"https://github.com/navervision/lincir"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-retrieval-on-fashion-iq","task":"Image Retrieval","dataset_variant":"Fashion IQ","rows":22,"metrics":["(Recall@10+Recall@50)/2","Recall@10"],"first_row_in_archive_order":{"model":"DQU-CIR","paper":null,"metrics":{"(Recall@10+Recall@50)/2":"71.77","Recall@10":"61.97"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-retrieval-with-multi-modal-query-on-1","task":"Image Retrieval with Multi-Modal Query","dataset_variant":"FashionIQ","rows":2,"metrics":["Recall@10"],"first_row_in_archive_order":{"model":"ComposeAE","paper":"/paper/compositional-learning-of-image-text-query","metrics":{"Recall@10":"11.8"},"code_links":[{"title":"ecom-research/ComposeAE","url":"https://github.com/ecom-research/ComposeAE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/composed-image-retrieval-coir-on-fashion-iq","task":"Composed Image Retrieval (CoIR)","dataset_variant":"Fashion IQ","rows":1,"metrics":["(Recall@10+Recall@50)/2","R@10","R@50"],"first_row_in_archive_order":{"model":"CoVR-BLIP-2","paper":"/paper/covr-learning-composed-video-retrieval-from","metrics":{"(Recall@10+Recall@50)/2":"60.57","R@10":"49.96","R@50":"71.17"},"code_links":[{"title":"lucas-ventura/CoVR","url":"https://github.com/lucas-ventura/CoVR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/virtual-try-on-on-fashioniq","task":"Virtual Try-on","dataset_variant":"FashionIQ","rows":1,"metrics":["10 fold Cross validation"],"first_row_in_archive_order":{"model":"mm","paper":"/paper/swapnet-garment-transfer-in-single-view","metrics":{"10 fold Cross validation":"66"},"code_links":[{"title":"andrewjong/SwapNet","url":"https://github.com/andrewjong/SwapNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-3","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset_variant":"FashionIQ","rows":1,"metrics":["R@10"],"first_row_in_archive_order":{"model":"SEARLE-XL-OTI","paper":"/paper/zero-shot-composed-image-retrieval-with","metrics":{"R@10":"27.61"},"code_links":[{"title":"miccunifi/searle","url":"https://github.com/miccunifi/searle"},{"title":"miccunifi/circo","url":"https://github.com/miccunifi/circo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/tmcir-token-merge-benefits-composed-image","title":"TMCIR: Token Merge Benefits Composed Image Retrieval","date":"2025-04-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/collm-a-large-language-model-for-composed","title":"CoLLM: A Large Language Model for Composed Image Retrieval","date":"2025-03-25","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/imagescope-unifying-language-guided-image-1","title":"ImageScope: Unifying Language-Guided Image Retrieval via Large Multimodal Model Collective Reasoning","date":"2025-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scot-self-supervised-contrastive-pretraining","title":"SCOT: Self-Supervised Contrastive Pretraining For Zero-Shot Compositional Retrieval","date":"2025-01-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/megapairs-massive-data-synthesis-for","title":"MegaPairs: Massive Data Synthesis For Universal Multimodal Retrieval","date":"2024-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/reason-before-retrieve-one-stage-reflective","title":"Reason-before-Retrieve: One-Stage Reflective Chain-of-Thoughts for Training-Free Zero-Shot Composed Image Retrieval","date":"2024-12-15","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/semantic-editing-increment-benefits-zero-shot","title":"Semantic Editing Increment Benefits Zero-Shot Composed Image Retrieval","date":"2024-10-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/training-free-zs-cir-via-weighted-modality","title":"Training-free Zero-shot Composed Image Retrieval via Weighted Modality Fusion and Similarity","date":"2024-09-07","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ldre-llm-based-divergent-reasoning-and","title":"LDRE: LLM-based Divergent Reasoning and Ensemble for Zero-Shot Composed Image Retrieval","date":"2024-07-11","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/reducing-task-discrepancy-of-text-encoders","title":"An Efficient Post-hoc Framework for Reducing Task Discrepancy of Text Encoders for Composed Image Retrieval","date":"2024-06-13","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/cala-complementary-association-learning-for","title":"CaLa: Complementary Association Learning for Augmenting Composed Image Retrieval","date":"2024-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-composed-image-retrieval-via","title":"Improving Composed Image Retrieval via Contrastive Learning with Scaling Positives and Negatives","date":"2024-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/magiclens-self-supervised-image-retrieval","title":"MagicLens: Self-Supervised Image Retrieval with Open-Ended Instructions","date":"2024-03-28","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-only-efficient-training-of-zero-shot","title":"Language-only Efficient Training of Zero-shot Composed Image Retrieval","date":"2023-12-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pretrain-like-you-inference-masked-tuning","title":"Pretrain like Your Inference: Masked Tuning Improves Zero-Shot Composed Image Retrieval","date":"2023-11-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-by-language-for-training-free","title":"Vision-by-Language for Training-Free Compositional Image Retrieval","date":"2023-10-13","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sentence-level-prompts-benefit-composed-image","title":"Sentence-level Prompts Benefit Composed Image Retrieval","date":"2023-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/composed-image-retrieval-using-contrastive","title":"Composed Image Retrieval using Contrastive Learning and Task-oriented CLIP-based Features","date":"2023-08-22","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ranking-aware-uncertainty-for-text-guided","title":"Ranking-aware Uncertainty for Text-guided Image Retrieval","date":"2023-08-16","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/zero-shot-composed-text-image-retrieval","title":"Zero-shot Composed Text-Image Retrieval","date":"2023-06-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/relieving-triplet-ambiguity-consensus-network","title":"Collaborative Group: Composed Image Retrieval via Consensus Learning from Noisy Annotations","date":"2023-06-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/candidate-set-re-ranking-for-composed-image","title":"Candidate Set Re-ranking for Composed Image Retrieval with Dual Multi-modal Encoder","date":"2023-05-25","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bi-directional-training-for-composed-image","title":"Bi-directional Training for Composed Image Retrieval via Text Prompt Learning","date":"2023-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/zero-shot-composed-image-retrieval-with","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","date":"2023-03-27","rows_on_this_dataset":5,"code_links":2,"syntology":null},{"paper":"/paper/compodiff-versatile-composed-image-retrieval","title":"CompoDiff: Versatile Composed Image Retrieval With Latent Diffusion","date":"2023-03-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/data-roaming-and-early-fusion-for-composed","title":"Data Roaming and Quality Assessment for Composed Image Retrieval","date":"2023-03-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","date":"2023-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/composed-image-retrieval-with-text-feedback","title":"Composed Image Retrieval with Text Feedback via Multi-grained Uncertainty Regularization","date":"2022-11-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/conditioned-and-composed-image-retrieval","title":"Conditioned and Composed Image Retrieval Combining and Partially Fine-Tuning CLIP-Based Features","date":"2022-06-19","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/this-is-my-unicorn-fluffy-personalizing","title":"\"This is my unicorn, Fluffy\": Personalizing frozen vision-language representations","date":"2022-04-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/effective-conditioned-and-composed-image","title":"Effective Conditioned and Composed Image Retrieval Combining CLIP-Based Features","date":"2022-01-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/cosmo-content-style-modulation-for-image","title":"CoSMo: Content-Style Modulation for Image Retrieval With Text Feedback","date":"2021-06-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rtic-residual-learning-for-text-and-image","title":"RTIC: Residual Learning for Text and Image Composition using Graph Convolutional Network","date":"2021-04-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/compositional-learning-of-image-text-query","title":"Compositional Learning of Image-Text Query for Image Retrieval","date":"2020-06-19","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/image-search-with-text-feedback-by","title":"Image Search With Text Feedback by Visiolinguistic Attention Learning","date":"2020-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/curlingnet-compositional-learning-between","title":"CurlingNet: Compositional Learning between Images and Text for Fashion IQ Data","date":"2020-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/composing-text-and-image-for-image-retrieval","title":"Composing Text and Image for Image Retrieval - An Empirical Odyssey","date":"2018-12-18","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/swapnet-garment-transfer-in-single-view","title":"SwapNet: Garment Transfer in Single View Images","date":"2018-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":18,"samples_harvested":72,"samples_ran":44,"samples_unverified":28,"pointer_only_for_licence":25,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}