{"url":"/sota/unsupervised-semantic-segmentation-with-11","task":{"name":"Unsupervised Semantic Segmentation with Language-image Pre-training","url":"/task/unsupervised-semantic-segmentation-with","note":null},"dataset":{"name":"PASCAL VOC","url":"/dataset/pascal-voc"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"A segmentation task which does not utilise any human-level supervision for semantic segmentation except for a backbone which is initialised with features pre-trained with image-level labels.","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["mIoU"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"mIoU":null}},"counts":{"rows":10,"rows_with_code":10,"rows_with_paper_page":10,"rows_dated":10,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"CorrCLIP","metrics":{"mIoU":"76.7"},"uses_additional_data":false,"paper_date":"2024-11-15","paper":"/paper/corrclip-reconstructing-correlations-in-clip","paper_url":"https://arxiv.org/abs/2411.10086v1","paper_title":"CorrCLIP: Reconstructing Correlations in CLIP with Off-the-Shelf Foundation Models for Open-Vocabulary Semantic Segmentation","code":"https://github.com/zdk258/CorrCLIP","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":9,"n_samples":9,"n_pointer_only_licence":9}},{"rank_in_archive_order":2,"model":"TextRegion","metrics":{"mIoU":"73.1"},"uses_additional_data":false,"paper_date":"2025-05-29","paper":"/paper/textregion-text-aligned-region-tokens-from","paper_url":"https://arxiv.org/abs/2505.23769v1","paper_title":"TextRegion: Text-Aligned Region Tokens from Frozen Image-Text Models","code":"https://github.com/avaxiao/TextRegion","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"Trident","metrics":{"mIoU":"70.8"},"uses_additional_data":false,"paper_date":"2024-11-14","paper":"/paper/harnessing-vision-foundation-models-for-high","paper_url":"https://arxiv.org/abs/2411.09219v1","paper_title":"Harnessing Vision Foundation Models for High-Performance, Training-Free Open Vocabulary Segmentation","code":"https://github.com/YuHengsss/Trident","n_code_links":1,"syntology":null},{"rank_in_archive_order":4,"model":"CLS-SEG","metrics":{"mIoU":"68.7"},"uses_additional_data":false,"paper_date":"2023-12-20","paper":"/paper/tagclip-a-local-to-global-framework-to","paper_url":"https://arxiv.org/abs/2312.12828v1","paper_title":"TagCLIP: A Local-to-Global Framework to Enhance Open-Vocabulary Multi-Label Classification of CLIP Without Training","code":"https://github.com/linyq2117/tagclip","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"ProxyCLIP","metrics":{"mIoU":"65.0"},"uses_additional_data":false,"paper_date":"2024-08-09","paper":"/paper/proxyclip-proxy-attention-improves-clip-for","paper_url":"https://arxiv.org/abs/2408.04883v1","paper_title":"ProxyCLIP: Proxy Attention Improves CLIP for Open-Vocabulary Segmentation","code":"https://github.com/mc-lan/proxyclip","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":6,"n_samples":9,"n_pointer_only_licence":9}},{"rank_in_archive_order":6,"model":"TTD (TCL)","metrics":{"mIoU":"61.1"},"uses_additional_data":false,"paper_date":"2024-03-30","paper":"/paper/ttd-text-tag-self-distillation-enhancing","paper_url":"https://arxiv.org/abs/2404.00384v2","paper_title":"TTD: Text-Tag Self-Distillation Enhancing Image-Text Alignment in CLIP to Alleviate Single Tag Bias","code":"https://github.com/shjo-april/TTD","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":7,"model":"TCL","metrics":{"mIoU":"55.0"},"uses_additional_data":false,"paper_date":"2022-12-01","paper":"/paper/learning-to-generate-text-grounded-mask-for","paper_url":"https://arxiv.org/abs/2212.00785v2","paper_title":"Learning to Generate Text-grounded Mask for Open-world Semantic Segmentation from Only Image-Text Pairs","code":"https://github.com/kakaobrain/tcl","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":7,"n_samples":8,"n_pointer_only_licence":0}},{"rank_in_archive_order":8,"model":"TagAlign","metrics":{"mIoU":"53.9"},"uses_additional_data":false,"paper_date":"2023-12-21","paper":"/paper/tagalign-improving-vision-language-alignment","paper_url":"https://arxiv.org/abs/2312.14149v4","paper_title":"TagAlign: Improving Vision-Language Alignment with Multi-Tag Classification","code":"https://github.com/Qinying-Liu/TagAlign","n_code_links":1,"syntology":null},{"rank_in_archive_order":9,"model":"TTD (MaskCLIP)","metrics":{"mIoU":"43.1"},"uses_additional_data":false,"paper_date":"2024-03-30","paper":"/paper/ttd-text-tag-self-distillation-enhancing","paper_url":"https://arxiv.org/abs/2404.00384v2","paper_title":"TTD: Text-Tag Self-Distillation Enhancing Image-Text Alignment in CLIP to Alleviate Single Tag Bias","code":"https://github.com/shjo-april/TTD","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":10,"model":"MaskCLIP","metrics":{"mIoU":"29.3"},"uses_additional_data":false,"paper_date":"2021-12-02","paper":"/paper/denseclip-extract-free-dense-labels-from-clip","paper_url":"https://arxiv.org/abs/2112.01071v2","paper_title":"Extract Free Dense Labels from CLIP","code":"https://github.com/chongzhou96/maskclip","n_code_links":1,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":5,"rows_with_any_sample_ran":4,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":3,"samples_over_distinct_papers":{"n_ran":5,"n_unverified":22,"n_samples":27,"n_pointer_only_licence":19,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":6,"n_unverified":22,"n_samples":28,"n_pointer_only_licence":20,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}