{"url":"/task/multi-modal","name":"Image Retrieval with Multi-Modal Query","slug":"multi-modal","description_markdown":"The problem of retrieving images from a database based on a multi-modal (image- text) query. Specifically, the query text prompts some modification in the query image and the task is to retrieve images with the desired modifications.","categories":[{"name":"Miscellaneous","url":"/area/miscellaneous"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":10,"papers_with_code":9,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":4,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/image-retrieval-with-multi-modal-query-on","slug":"image-retrieval-with-multi-modal-query-on","dataset":"Fashion200k","dataset_url":null,"rows_in_archive":8,"metrics":["Recall@1","Recall@10","Recall@50"],"first_row_in_archive_order":{"model":"Css-Net","paper_title":"Collaborative Group: Composed Image Retrieval via Consensus Learning from Noisy Annotations","paper_url":"/paper/relieving-triplet-ambiguity-consensus-network","paper_date":"2023-06-03","arxiv_id":"2306.02092","code_links":[],"syntology":null}},{"leaderboard":"/sota/image-retrieval-with-multi-modal-query-on-mit","slug":"image-retrieval-with-multi-modal-query-on-mit","dataset":"MIT-States","dataset_url":"/dataset/mit-states","rows_in_archive":5,"metrics":["Recall@1","Recall@5","Recall@10"],"first_row_in_archive_order":{"model":"ComposeAE","paper_title":"Compositional Learning of Image-Text Query for Image Retrieval","paper_url":"/paper/compositional-learning-of-image-text-query","paper_date":"2020-06-19","arxiv_id":"2006.11149","code_links":[{"title":"ecom-research/ComposeAE","url":"https://github.com/ecom-research/ComposeAE"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/image-retrieval-with-multi-modal-query-on-1","slug":"image-retrieval-with-multi-modal-query-on-1","dataset":"FashionIQ","dataset_url":"/dataset/fashion-iq","rows_in_archive":2,"metrics":["Recall@10"],"first_row_in_archive_order":{"model":"ComposeAE","paper_title":"Compositional Learning of Image-Text Query for Image Retrieval","paper_url":"/paper/compositional-learning-of-image-text-query","paper_date":"2020-06-19","arxiv_id":"2006.11149","code_links":[{"title":"ecom-research/ComposeAE","url":"https://github.com/ecom-research/ComposeAE"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/fashion-iq","name":"Fashion IQ","full_name":"","num_papers_in_archive":102},{"url":"/dataset/mit-states","name":"MIT-States","full_name":"","num_papers_in_archive":91}],"subtasks":[{"url":"/task/cross-modal-information-retrieval","name":"Cross-Modal Information Retrieval"},{"url":"/task/cross-modal-retrieval","name":"Cross-Modal Retrieval"},{"url":"/task/multi-modal-person-identification","name":"Multi-Modal Person Identification"},{"url":"/task/zero-shot-cross-modal-retrieval","name":"Zero-Shot Cross-Modal Retrieval"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":9,"of":9,"tagged_in_all":10,"items":[{"url":"/paper/show-and-tell-a-neural-image-caption","title":"Show and Tell: A Neural Image Caption Generator","date":"2014-11-17","arxiv_id":"1411.4555","repositories_listed":74,"syntology":{"n":34,"n_ran":13,"n_unverified":21,"n_pointer_only":6}},{"url":"/paper/a-simple-neural-network-module-for-relational","title":"A simple neural network module for relational reasoning","date":"2017-06-05","arxiv_id":"1706.01427","repositories_listed":20,"syntology":{"n":7,"n_ran":3,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/film-visual-reasoning-with-a-general","title":"FiLM: Visual Reasoning with a General Conditioning Layer","date":"2017-09-22","arxiv_id":"1709.07871","repositories_listed":7,"syntology":{"n":10,"n_ran":9,"n_unverified":1,"n_pointer_only":7}},{"url":"/paper/composing-text-and-image-for-image-retrieval","title":"Composing Text and Image for Image Retrieval - An Empirical Odyssey","date":"2018-12-18","arxiv_id":"1812.07119","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/composed-image-retrieval-with-text-feedback","title":"Composed Image Retrieval with Text Feedback via Multi-grained Uncertainty Regularization","date":"2022-11-14","arxiv_id":"2211.07394","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/compositional-learning-of-image-text-query","title":"Compositional Learning of Image-Text Query for Image Retrieval","date":"2020-06-19","arxiv_id":"2006.11149","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/attributes-as-operators-factorizing-unseen","title":"Attributes as Operators: Factorizing Unseen Attribute-Object Compositions","date":"2018-03-27","arxiv_id":"1803.09851","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/automatic-spatially-aware-fashion-concept","title":"Automatic Spatially-aware Fashion Concept Discovery","date":"2017-08-03","arxiv_id":"1708.01311","repositories_listed":1,"syntology":null},{"url":"/paper/image-question-answering-using-convolutional","title":"Image Question Answering using Convolutional Neural Network with Dynamic Parameter Prediction","date":"2015-11-18","arxiv_id":"1511.05756","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}