{"url":"/dataset/refmatte","name":"RefMatte","full_name":"Referring Image Matting","description_markdown":"RefMatte is the first large-scale challenging dataset under the task referring image matting, generated by a comprehensive image composition and expression generation engine on top of current public high-quality matting foregrounds with flexible logics and re-labelled diverse attributes.  RefMatte consists of 230 object categories, 47,500 images, 118,749 expression-region entities, and 474,996 expressions, which can be further extended easily in the future.\r\n\r\nRefMatte comes along with two settings: keyword-based and expression-based. The former one takes a high-resolution image and a keyword as input, while the latter one takes a high-resolution image and a flowery expression as input.\r\n\r\nAdditionally, we construct a real-world test set with 100 high-resolution natural images and manually annotate complex phrases to evaluate the out-of-domain generalization abilities of RIM methods, named as RefMatte-RW100.","description_withheld":null,"homepage":"https://github.com/jizhiziLi/rim","introduced_date":"2022-06-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/referring-image-matting","title":"Referring Image Matting","first_author":"Jizhizi Li","url":null},"license":{"name":"CC BY-NC","url":"https://creativecommons.org/licenses/by-nc/4.0/"},"modalities":[],"tasks":[{"name":"Referring Image Matting (Expression-based)","url":"/task/referring-image-matting-expression-based","datasets_with_task":"/datasets/task/referring-image-matting-expression-based"},{"name":"Referring Image Matting (Keyword-based)","url":"/task/referring-image-matting-keyword-based","datasets_with_task":"/datasets/task/referring-image-matting-keyword-based"},{"name":"Referring Image Matting (RefMatte-RW100)","url":"/task/referring-image-matting-refmatte-rw100","datasets_with_task":"/datasets/task/referring-image-matting-refmatte-rw100"},{"name":"Referring Image Matting (Prompt-based)","url":"/task/referring-image-matting-prompt-based","datasets_with_task":"/datasets/task/referring-image-matting-prompt-based"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["RefMatte"],"data_loaders":[],"num_papers_in_archive":7,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/referring-image-matting-expression-based-on","task":"Referring Image Matting (Expression-based)","dataset_variant":"RefMatte","rows":4,"metrics":["SAD","MSE","MAD","SAD(E)","MSE(E)","MAD(E)"],"first_row_in_archive_order":{"model":"CLIPMat (ViT-L/14)","paper":"/paper/referring-image-matting","metrics":{"MAD":"0.0238","MAD(E)":"0.0254","MSE":"0.0212","MSE(E)":"0.0226","SAD":"42.05","SAD(E)":"44.77"},"code_links":[{"title":"jizhizili/rim","url":"https://github.com/jizhizili/rim"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/referring-image-matting-keyword-based-on","task":"Referring Image Matting (Keyword-based)","dataset_variant":"RefMatte","rows":4,"metrics":["SAD","MSE","MAD","SAD(E)","MSE(E)","MAD(E)"],"first_row_in_archive_order":{"model":"CLIPMat (ViT-L/14)","paper":"/paper/referring-image-matting","metrics":{"MAD":"0.0049","MAD(E)":"0.0051","MSE":"0.0022","MSE(E)":"0.0023","SAD":"8.51","SAD(E)":"8.98"},"code_links":[{"title":"jizhizili/rim","url":"https://github.com/jizhizili/rim"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/referring-image-matting-refmatte-rw100-on","task":"Referring Image Matting (RefMatte-RW100)","dataset_variant":"RefMatte","rows":4,"metrics":["SAD","MSE","MAD","SAD(E)","MSE(E)","MAD(E)"],"first_row_in_archive_order":{"model":"CLIPMat (ViT-L/14)","paper":"/paper/referring-image-matting","metrics":{"MAD":"0.0510","MAD(E)":"0.0505","MSE":"0.0488","MSE(E)":"0.0483","SAD":"88.52","SAD(E)":"87.92"},"code_links":[{"title":"jizhizili/rim","url":"https://github.com/jizhizili/rim"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/referring-image-matting","title":"Referring Image Matting","date":"2022-06-10","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/prompt-based-multi-modal-image-segmentation","title":"Image Segmentation Using Text and Image Prompts","date":"2021-12-18","rows_on_this_dataset":3,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":13,"samples_ran":7,"samples_unverified":6,"pointer_only_for_licence":2,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}