{"url":"/sota/unsupervised-semantic-segmentation-with-7","task":{"name":"Unsupervised Semantic Segmentation with Language-image Pre-training","url":"/task/unsupervised-semantic-segmentation-with","note":null},"dataset":{"name":"PascalVOC-20","url":"/dataset/pascal-voc"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"A segmentation task which does not utilise any human-level supervision for semantic segmentation except for a backbone which is initialised with features pre-trained with image-level labels.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["mIoU"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"mIoU":null}},"counts":{"rows":10,"rows_with_code":10,"rows_with_paper_page":10,"rows_dated":10,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"CorrCLIP","metrics":{"mIoU":"91.8"},"uses_additional_data":false,"paper_date":"2024-11-15","paper":"/paper/corrclip-reconstructing-correlations-in-clip","paper_url":"https://arxiv.org/abs/2411.10086v1","paper_title":"CorrCLIP: Reconstructing Correlations in CLIP with Off-the-Shelf Foundation Models for Open-Vocabulary Semantic Segmentation","code":"https://github.com/zdk258/CorrCLIP","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":9,"n_samples":9,"n_pointer_only_licence":9}},{"rank_in_archive_order":2,"model":"TextRegion","metrics":{"mIoU":"89.5"},"uses_additional_data":false,"paper_date":"2025-05-29","paper":"/paper/textregion-text-aligned-region-tokens-from","paper_url":"https://arxiv.org/abs/2505.23769v1","paper_title":"TextRegion: Text-Aligned Region Tokens from Frozen Image-Text Models","code":"https://github.com/avaxiao/TextRegion","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"Trident","metrics":{"mIoU":"88.7"},"uses_additional_data":false,"paper_date":"2024-11-14","paper":"/paper/harnessing-vision-foundation-models-for-high","paper_url":"https://arxiv.org/abs/2411.09219v1","paper_title":"Harnessing Vision Foundation Models for High-Performance, Training-Free Open Vocabulary Segmentation","code":"https://github.com/YuHengsss/Trident","n_code_links":1,"syntology":null},{"rank_in_archive_order":4,"model":"TagAlign","metrics":{"mIoU":"87.9"},"uses_additional_data":false,"paper_date":"2023-12-21","paper":"/paper/tagalign-improving-vision-language-alignment","paper_url":"https://arxiv.org/abs/2312.14149v4","paper_title":"TagAlign: Improving Vision-Language Alignment with Multi-Tag Classification","code":"https://github.com/Qinying-Liu/TagAlign","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"ProxyCLIP","metrics":{"mIoU":"83.3"},"uses_additional_data":false,"paper_date":"2024-08-09","paper":"/paper/proxyclip-proxy-attention-improves-clip-for","paper_url":"https://arxiv.org/abs/2408.04883v1","paper_title":"ProxyCLIP: Proxy Attention Improves CLIP for Open-Vocabulary Segmentation","code":"https://github.com/mc-lan/proxyclip","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":6,"n_samples":9,"n_pointer_only_licence":9}},{"rank_in_archive_order":6,"model":"TCL","metrics":{"mIoU":"83.2"},"uses_additional_data":false,"paper_date":"2022-12-01","paper":"/paper/learning-to-generate-text-grounded-mask-for","paper_url":"https://arxiv.org/abs/2212.00785v2","paper_title":"Learning to Generate Text-grounded Mask for Open-world Semantic Segmentation from Only Image-Text Pairs","code":"https://github.com/kakaobrain/tcl","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":7,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"GroupViT (RedCaps)","metrics":{"mIoU":"79.7"},"uses_additional_data":false,"paper_date":"2022-02-22","paper":"/paper/groupvit-semantic-segmentation-emerges-from","paper_url":"https://arxiv.org/abs/2202.11094v5","paper_title":"GroupViT: Semantic Segmentation Emerges from Text Supervision","code":"https://github.com/huggingface/transformers","n_code_links":6,"syntology":null},{"rank_in_archive_order":8,"model":"COSMOS ViT-B/16","metrics":{"mIoU":"77.7"},"uses_additional_data":false,"paper_date":"2024-12-02","paper":"/paper/cosmos-cross-modality-self-distillation-for","paper_url":"https://arxiv.org/abs/2412.01814v2","paper_title":"COSMOS: Cross-Modality Self-Distillation for Vision Language Pre-training","code":"https://github.com/ExplainableML/cosmos","n_code_links":1,"syntology":null},{"rank_in_archive_order":9,"model":"MaskCLIP","metrics":{"mIoU":"74.9"},"uses_additional_data":false,"paper_date":"2021-12-02","paper":"/paper/denseclip-extract-free-dense-labels-from-clip","paper_url":"https://arxiv.org/abs/2112.01071v2","paper_title":"Extract Free Dense Labels from CLIP","code":"https://github.com/chongzhou96/maskclip","n_code_links":1,"syntology":null},{"rank_in_archive_order":10,"model":"ReCo","metrics":{"mIoU":"57.7"},"uses_additional_data":false,"paper_date":"2022-06-14","paper":"/paper/reco-retrieve-and-co-segment-for-zero-shot-1","paper_url":"https://arxiv.org/abs/2206.07045v1","paper_title":"ReCo: Retrieve and Co-segment for Zero-shot Transfer","code":"https://github.com/NoelShin/reco","n_code_links":2,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":3,"rows_with_any_sample_ran":2,"distinct_papers_with_graph_line":3,"distinct_papers_with_any_sample_ran":2,"samples_over_distinct_papers":{"n_ran":4,"n_unverified":22,"n_samples":26,"n_pointer_only_licence":18,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":4,"n_unverified":22,"n_samples":26,"n_pointer_only_licence":18,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}