{"url":"/dataset/mit-states","name":"MIT-States","full_name":null,"description_markdown":"The **MIT-States** dataset has 245 object classes, 115 attribute classes and ∼53K images. There is a wide range of objects (e.g., fish, persimmon, room) and attributes (e.g., mossy, deflated, dirty). On average, each object instance is modified by one of the 9 attributes it affords.\r\n\r\nSource: [Attributes as Operators: Factorizing Unseen Attribute-Object Compositions](https://arxiv.org/abs/1803.09851)\r\nImage Source: [http://web.mit.edu/phillipi/Public/states_and_transformations/index.html](http://web.mit.edu/phillipi/Public/states_and_transformations/index.html)","description_withheld":null,"homepage":"http://web.mit.edu/phillipi/Public/states_and_transformations/index.html","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/discovering-states-and-transformations-in","title":"Discovering States and Transformations in Image Collections","first_author":"Phillip Isola","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Image Retrieval with Multi-Modal Query","url":"/task/multi-modal","datasets_with_task":"/datasets/task/multi-modal"},{"name":"Compositional Zero-Shot Learning","url":"/task/compositional-zero-shot-learning","datasets_with_task":"/datasets/task/compositional-zero-shot-learning"}],"languages":[],"variants":["MIT-States","MIT-States, generalized split"],"data_loaders":[],"num_papers_in_archive":91,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/image-retrieval-with-multi-modal-query-on-mit","task":"Image Retrieval with Multi-Modal Query","dataset_variant":"MIT-States","rows":5,"metrics":["Recall@1","Recall@5","Recall@10"],"first_row_in_archive_order":{"model":"ComposeAE","paper":"/paper/compositional-learning-of-image-text-query","metrics":{"Recall@1":"13.9","Recall@10":"47.9","Recall@5":"35.5"},"code_links":[{"title":"ecom-research/ComposeAE","url":"https://github.com/ecom-research/ComposeAE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/compositional-zero-shot-learning-on-mit-2","task":"Compositional Zero-Shot Learning","dataset_variant":"MIT-States","rows":2,"metrics":["AUC","Attribute accuracy","Object accuracy","Seen accuracy","Top-1 accuracy %","Top-2 accuracy %","Top-3 accuracy %","Unseen accuracy","best HM"],"first_row_in_archive_order":{"model":"CANet","paper":"/paper/learning-conditional-attributes-for-1","metrics":{"AUC":"5.4","Attribute accuracy":"30.2","Object accuracy":"32.6","Seen accuracy":"29","Unseen accuracy":"26.2","best HM":"17.9"},"code_links":[{"title":"wqshmzh/canet-czsl","url":"https://github.com/wqshmzh/canet-czsl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/compositional-zero-shot-learning-on-mit-3","task":"Compositional Zero-Shot Learning","dataset_variant":"MIT-States, generalized split","rows":2,"metrics":["H-Mean","Seen accuracy","Test AUC top 1","Test AUC top 2","Test AUC top 3","Unseen accuracy","Val AUC top 1","Val AUC top 2","Val AUC top 3"],"first_row_in_archive_order":{"model":"CAILA","paper":"/paper/caila-concept-aware-intra-layer-adapters-for","metrics":{"H-Mean":"39.9","Seen accuracy":"51.0","Test AUC top 1":"23.4","Test AUC top 2":"-","Test AUC top 3":"-","Unseen accuracy":"53.9","Val AUC top 1":"-","Val AUC top 2":"-","Val AUC top 3":"-"},"code_links":[{"title":"zhaohengz/llamp","url":"https://github.com/zhaohengz/llamp"},{"title":"zhaohengz/caila","url":"https://github.com/zhaohengz/caila"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-mit-states-1","task":"Zero-Shot Learning","dataset_variant":"MIT-States","rows":1,"metrics":["A-acc"],"first_row_in_archive_order":{"model":"CZSL","paper":"/paper/locl-learning-object-attribute-composition","metrics":{"A-acc":"36.0"},"code_links":[{"title":"satish1901/LOCL-Learning-Object-Attribute-Composition-using-Localization","url":"https://github.com/satish1901/LOCL-Learning-Object-Attribute-Composition-using-Localization"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/learning-conditional-attributes-for-1","title":"Learning Conditional Attributes for Compositional Zero-Shot Learning","date":"2023-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/caila-concept-aware-intra-layer-adapters-for","title":"CAILA: Concept-Aware Intra-Layer Adapters for Compositional Zero-Shot Learning","date":"2023-05-26","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/locl-learning-object-attribute-composition","title":"LOCL: Learning Object-Attribute Composition using Localization","date":"2022-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/compositional-learning-of-image-text-query","title":"Compositional Learning of Image-Text Query for Image Retrieval","date":"2020-06-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/symmetry-and-group-in-attribute-object","title":"Symmetry and Group in Attribute-Object Compositions","date":"2020-04-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/composing-text-and-image-for-image-retrieval","title":"Composing Text and Image for Image Retrieval - An Empirical Odyssey","date":"2018-12-18","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attributes-as-operators-factorizing-unseen","title":"Attributes as Operators: Factorizing Unseen Attribute-Object Compositions","date":"2018-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/film-visual-reasoning-with-a-general","title":"FiLM: Visual Reasoning with a General Conditioning Layer","date":"2017-09-22","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":9,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/show-and-tell-a-neural-image-caption","title":"Show and Tell: A Neural Image Caption Generator","date":"2014-11-17","rows_on_this_dataset":1,"code_links":74,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":13,"samples_unverified":21,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":58,"samples_ran":28,"samples_unverified":30,"pointer_only_for_licence":22,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}