{"url":"/dataset/ade20k","name":"ADE20K","full_name":null,"description_markdown":"The **ADE20K** semantic segmentation dataset contains more than 20K scene-centric images exhaustively annotated with pixel-level objects and object parts labels. There are totally 150 semantic categories, which include stuffs like sky, road, grass, and discrete objects like person, car, bed.\r\n\r\nSource: [Cooperative Image Segmentation and Restoration in Adverse Environmental Conditions](https://arxiv.org/abs/1911.00679)\r\nImage Source: [https://groups.csail.mit.edu/vision/datasets/ADE20K/](https://groups.csail.mit.edu/vision/datasets/ADE20K/)","description_withheld":null,"homepage":"https://groups.csail.mit.edu/vision/datasets/ADE20K/","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/scene-parsing-through-ade20k-dataset","title":"Scene Parsing Through ADE20K Dataset","first_author":"Bolei Zhou","url":null},"license":{"name":"Custom (research-only, non-commercial)","url":"https://groups.csail.mit.edu/vision/datasets/ADE20K/terms/"},"modalities":[],"tasks":[{"name":"Semantic Segmentation","url":"/task/semantic-segmentation","datasets_with_task":"/datasets/task/semantic-segmentation"},{"name":"Instance Segmentation","url":"/task/instance-segmentation","datasets_with_task":"/datasets/task/instance-segmentation"},{"name":"Image-to-Image Translation","url":"/task/image-to-image-translation","datasets_with_task":"/datasets/task/image-to-image-translation"},{"name":"Semi-Supervised Semantic Segmentation","url":"/task/semi-supervised-semantic-segmentation","datasets_with_task":"/datasets/task/semi-supervised-semantic-segmentation"},{"name":"Panoptic Segmentation","url":"/task/panoptic-segmentation","datasets_with_task":"/datasets/task/panoptic-segmentation"},{"name":"Scene Understanding","url":"/task/scene-understanding","datasets_with_task":"/datasets/task/scene-understanding"},{"name":"Reconstruction","url":"/task/reconstruction","datasets_with_task":"/datasets/task/reconstruction"},{"name":"Face Detection","url":"/task/face-detection","datasets_with_task":"/datasets/task/face-detection"},{"name":"Open Vocabulary Semantic Segmentation","url":"/task/open-vocabulary-semantic-segmentation","datasets_with_task":"/datasets/task/open-vocabulary-semantic-segmentation"},{"name":"Unsupervised Semantic Segmentation with Language-image Pre-training","url":"/task/unsupervised-semantic-segmentation-with","datasets_with_task":"/datasets/task/unsupervised-semantic-segmentation-with"},{"name":"Weakly-Supervised Semantic Segmentation","url":"/task/weakly-supervised-semantic-segmentation","datasets_with_task":"/datasets/task/weakly-supervised-semantic-segmentation"},{"name":"Scene Recognition","url":"/task/scene-recognition","datasets_with_task":"/datasets/task/scene-recognition"},{"name":"Pose Transfer","url":"/task/pose-transfer","datasets_with_task":"/datasets/task/pose-transfer"},{"name":"Semi-Supervised Instance Segmentation","url":"/task/semi-supervised-instance-segmentation","datasets_with_task":"/datasets/task/semi-supervised-instance-segmentation"},{"name":"Continual Semantic Segmentation","url":"/task/continual-semantic-segmentation","datasets_with_task":"/datasets/task/continual-semantic-segmentation"},{"name":"Zero-Shot Semantic Segmentation","url":"/task/zero-shot-semantic-segmentation","datasets_with_task":"/datasets/task/zero-shot-semantic-segmentation"},{"name":"Overlapped 100-50","url":"/task/overlapped-100-50","datasets_with_task":"/datasets/task/overlapped-100-50"},{"name":"Overlapped 50-50","url":"/task/overlapped-50-50","datasets_with_task":"/datasets/task/overlapped-50-50"},{"name":"Overlapped 100-10","url":"/task/overlapped-100-10","datasets_with_task":"/datasets/task/overlapped-100-10"},{"name":"Overlapped 100-5","url":"/task/overlapped-100-5","datasets_with_task":"/datasets/task/overlapped-100-5"},{"name":"Overlapped 25-25","url":"/task/overlapped-25-25","datasets_with_task":"/datasets/task/overlapped-25-25"},{"name":"Open Vocabulary Panoptic Segmentation","url":"/task/open-vocabulary-panoptic-segmentation","datasets_with_task":"/datasets/task/open-vocabulary-panoptic-segmentation"},{"name":"Speech Prompted Semantic Segmentation","url":"/task/speech-prompted-semantic-segmentation","datasets_with_task":"/datasets/task/speech-prompted-semantic-segmentation"},{"name":"Sound Prompted Semantic Segmentation","url":"/task/sound-prompted-semantic-segmentation","datasets_with_task":"/datasets/task/sound-prompted-semantic-segmentation"},{"name":"Open-Vocabulary Semantic Segmentation","url":"/task/open-vocabulary-semantic-segmentation-1","datasets_with_task":"/datasets/task/open-vocabulary-semantic-segmentation-1"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["ADE20K-847","ADE20K-150","ADE20K 1/32 labeled","ADE20K 1/16 labeled","ADE20K-Outdoor Labels-to-Photos","ADE20K val","ADE20K Labels-to-Photos","ADE20K"],"data_loaders":[{"repo":"https://github.com/facebookresearch/detectron2","url":"https://detectron2.readthedocs.io/en/latest/tutorials/builtin_datasets.html#expected-dataset-structure-for-ade20k-scene-parsing","frameworks":["pytorch"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Xkull/phao_resize","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Xkull/ade_resize","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Xkull/ade","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Xkull/Phao_resize","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/scene_parse_150","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/zhoubolei/scene_parse_150","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/1aurent/ADE20K","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Xkull/phao_dataset","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Xkull/Phao_json","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Xkull/ade20k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/open-mmlab/mmsegmentation","url":"https://github.com/open-mmlab/mmsegmentation/blob/master/docs/dataset_prepare.md","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/scene_parse150","frameworks":["tf","jax"]},{"repo":"https://github.com/facebookresearch/MaskFormer","url":"https://github.com/facebookresearch/MaskFormer","frameworks":["pytorch"]}],"num_papers_in_archive":1213,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/semantic-segmentation-on-ade20k","task":"Semantic Segmentation","dataset_variant":"ADE20K","rows":235,"metrics":["Validation mIoU","Test Score","Params (M)","GFLOPs (512 x 512)","GFLOPs","Mean IoU (class)"],"first_row_in_archive_order":{"model":"ViT-P (InternImage-H)","paper":"/paper/the-missing-point-in-vision-transformers-for","metrics":{"Params (M)":"1610","Validation mIoU":"63.6"},"code_links":[{"title":"sajjad-sh33/vit-p","url":"https://github.com/sajjad-sh33/vit-p"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-segmentation-on-ade20k-val","task":"Semantic Segmentation","dataset_variant":"ADE20K val","rows":95,"metrics":["mIoU","Pixel Accuracy"],"first_row_in_archive_order":{"model":"BEiT-3","paper":"/paper/image-as-a-foreign-language-beit-pretraining","metrics":{"mIoU":"62.8"},"code_links":[{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm/tree/master/beit"},{"title":"lyan62/data-curation","url":"https://github.com/lyan62/data-curation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-ade20k-val","task":"Panoptic Segmentation","dataset_variant":"ADE20K val","rows":25,"metrics":["PQ","mIoU","AP"],"first_row_in_archive_order":{"model":"OneFormer (InternImage-H, emb_dim=256, single-scale, 896x896)","paper":"/paper/oneformer-one-transformer-to-rule-universal","metrics":{"AP":"40.2","PQ":"54.5","mIoU":"60.4"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"SHI-Labs/OneFormer","url":"https://github.com/SHI-Labs/OneFormer"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-1/oneformer"},{"title":"MindCode-4/code-2","url":"https://github.com/MindCode-4/code-2/tree/main/oneformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-semantic-segmentation-on-2","task":"Open Vocabulary Semantic Segmentation","dataset_variant":"ADE20K-150","rows":23,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"Mask-Adapter","paper":"/paper/mask-adapter-the-devil-is-in-the-masks-for","metrics":{"mIoU":"38.2"},"code_links":[{"title":"hustvl/maskadapter","url":"https://github.com/hustvl/maskadapter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-semantic-segmentation-on-3","task":"Open Vocabulary Semantic Segmentation","dataset_variant":"ADE20K-847","rows":19,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"UMG-CLIP-E/14","paper":"/paper/umg-clip-a-unified-multi-granularity-vision","metrics":{"mIoU":"17.3"},"code_links":[{"title":"lygsbw/umg-clip","url":"https://github.com/lygsbw/umg-clip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-to-image-translation-on-ade20k-labels","task":"Image-to-Image Translation","dataset_variant":"ADE20K Labels-to-Photos","rows":16,"metrics":["mIoU","Accuracy","FID","LPIPS"],"first_row_in_archive_order":{"model":"DP-SIMS (ConvNext-L)","paper":"/paper/unlocking-pre-trained-image-backbones-for","metrics":{"FID":"22.7","mIoU":"54.3"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/instance-segmentation-on-ade20k-val","task":"Instance Segmentation","dataset_variant":"ADE20K val","rows":14,"metrics":["AP","APL","APM","APS"],"first_row_in_archive_order":{"model":"OneFormer (InternImage-H, emb_dim=1024, single-scale, 896x896, COCO-Pretrained)","paper":"/paper/oneformer-one-transformer-to-rule-universal","metrics":{"AP":"44.2","APL":"64.3","APM":"49.9","APS":"23.7"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"SHI-Labs/OneFormer","url":"https://github.com/SHI-Labs/OneFormer"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-1/oneformer"},{"title":"MindCode-4/code-2","url":"https://github.com/MindCode-4/code-2/tree/main/oneformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-semantic-segmentation-with-4","task":"Unsupervised Semantic Segmentation with Language-image Pre-training","dataset_variant":"ADE20K","rows":13,"metrics":["Mean IoU (val)"],"first_row_in_archive_order":{"model":"CorrCLIP","paper":"/paper/corrclip-reconstructing-correlations-in-clip","metrics":{"Mean IoU (val)":"30.7"},"code_links":[{"title":"zdk258/CorrCLIP","url":"https://github.com/zdk258/CorrCLIP"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-panoptic-segmentation-on","task":"Open Vocabulary Panoptic Segmentation","dataset_variant":"ADE20K","rows":10,"metrics":["PQ"],"first_row_in_archive_order":{"model":"UMG-CLIP-E/14","paper":"/paper/umg-clip-a-unified-multi-granularity-vision","metrics":{"PQ":"31.6"},"code_links":[{"title":"lygsbw/umg-clip","url":"https://github.com/lygsbw/umg-clip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/overlapped-100-5-on-ade20k","task":"Overlapped 100-5","dataset_variant":"ADE20K","rows":8,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"MBS","paper":"/paper/mitigating-background-shift-in-class","metrics":{"mIoU":"42.8"},"code_links":[{"title":"roadonep/eccv2024_mbs","url":"https://github.com/roadonep/eccv2024_mbs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-to-image-translation-on-ade20k-outdoor","task":"Image-to-Image Translation","dataset_variant":"ADE20K-Outdoor Labels-to-Photos","rows":7,"metrics":["mIoU","Accuracy","FID"],"first_row_in_archive_order":{"model":"DP-GAN","paper":"/paper/dual-pyramid-generative-adversarial-networks","metrics":{"FID":"45.8","mIoU":"40.4"},"code_links":[{"title":"sj-li/dp_gan","url":"https://github.com/sj-li/dp_gan"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/overlapped-100-50-on-ade20k","task":"Overlapped 100-50","dataset_variant":"ADE20K","rows":7,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"MBS","paper":"/paper/mitigating-background-shift-in-class","metrics":{"mIoU":"45.7"},"code_links":[{"title":"roadonep/eccv2024_mbs","url":"https://github.com/roadonep/eccv2024_mbs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/overlapped-50-50-on-ade20k","task":"Overlapped 50-50","dataset_variant":"ADE20K","rows":7,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"MBS","paper":"/paper/mitigating-background-shift-in-class","metrics":{"mIoU":"45.4"},"code_links":[{"title":"roadonep/eccv2024_mbs","url":"https://github.com/roadonep/eccv2024_mbs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/overlapped-100-10-on-ade20k","task":"Overlapped 100-10","dataset_variant":"ADE20K","rows":6,"metrics":["Mean IoU (test) "],"first_row_in_archive_order":{"model":"MBS","paper":"/paper/mitigating-background-shift-in-class","metrics":{"Mean IoU (test) ":"44.5"},"code_links":[{"title":"roadonep/eccv2024_mbs","url":"https://github.com/roadonep/eccv2024_mbs"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semi-supervised-semantic-segmentation-on-41","task":"Semi-Supervised Semantic Segmentation","dataset_variant":"ADE20K 1/32 labeled","rows":5,"metrics":["Validation mIoU"],"first_row_in_archive_order":{"model":"UniMatch V2","paper":"/paper/unimatch-v2-pushing-the-limit-of-semi","metrics":{"Validation mIoU":"45.0"},"code_links":[{"title":"LiheYoung/UniMatch-V2","url":"https://github.com/LiheYoung/UniMatch-V2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semi-supervised-semantic-segmentation-on-42","task":"Semi-Supervised Semantic Segmentation","dataset_variant":"ADE20K 1/16 labeled","rows":5,"metrics":["Validation mIoU"],"first_row_in_archive_order":{"model":"UniMatch V2","paper":"/paper/unimatch-v2-pushing-the-limit-of-semi","metrics":{"Validation mIoU":"46.7"},"code_links":[{"title":"LiheYoung/UniMatch-V2","url":"https://github.com/LiheYoung/UniMatch-V2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sound-prompted-semantic-segmentation-on","task":"Sound Prompted Semantic Segmentation","dataset_variant":"ADE20K","rows":4,"metrics":["mAP","mIoU"],"first_row_in_archive_order":{"model":"DenseAV","paper":"/paper/separating-the-chirp-from-the-chat-self","metrics":{"mAP":"32.7","mIoU":"24.7"},"code_links":[{"title":"mhamilton723/DenseAV","url":"https://github.com/mhamilton723/DenseAV"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-prompted-semantic-segmentation-on","task":"Speech Prompted Semantic Segmentation","dataset_variant":"ADE20K","rows":4,"metrics":["mAP","mIoU"],"first_row_in_archive_order":{"model":"DenseAV","paper":"/paper/separating-the-chirp-from-the-chat-self","metrics":{"mAP":"48.7","mIoU":"36.8"},"code_links":[{"title":"mhamilton723/DenseAV","url":"https://github.com/mhamilton723/DenseAV"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/continual-semantic-segmentation-on-ade20k","task":"Continual Semantic Segmentation","dataset_variant":"ADE20K","rows":2,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"LGKD","paper":"/paper/label-guided-knowledge-distillation-for","metrics":{"mIoU":"37.5"},"code_links":[{"title":"ze-yang/lgkd","url":"https://github.com/ze-yang/lgkd"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/face-detection-on-ade20k","task":"Face Detection","dataset_variant":"ADE20K","rows":1,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"CASSOD","paper":"/paper/cassod-net-cascaded-and-separable-structures","metrics":{"mIoU":"42.86"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/overlapped-25-25-on-ade20k","task":"Overlapped 25-25","dataset_variant":"ADE20K","rows":1,"metrics":["Mean IoU (test)"],"first_row_in_archive_order":{"model":"SATS-M","paper":"/paper/sats-self-attention-transfer-for-continual","metrics":{"Mean IoU (test)":"32.56"},"code_links":[{"title":"QIU023/SATS_Continual_Semantic_Seg","url":"https://github.com/QIU023/SATS_Continual_Semantic_Seg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-ade20k","task":"Panoptic Segmentation","dataset_variant":"ADE20K","rows":1,"metrics":["PQ"],"first_row_in_archive_order":{"model":"MasQCLIP","paper":"/paper/masqclip-for-open-vocabulary-universal-image","metrics":{"PQ":"23.3"},"code_links":[{"title":"mlpc-ucsd/MasQCLIP","url":"https://github.com/mlpc-ucsd/MasQCLIP"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-transfer-on-ade20k","task":"Pose Transfer","dataset_variant":"ADE20K","rows":1,"metrics":["FID"],"first_row_in_archive_order":{"model":"SCAM","paper":"/paper/scam-transferring-humans-between-images-with","metrics":{"FID":"27.5"},"code_links":[{"title":"nicolas-dufour/SCAM","url":"https://github.com/nicolas-dufour/SCAM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/reconstruction-on-ade20k","task":"Reconstruction","dataset_variant":"ADE20K","rows":1,"metrics":["PSNR"],"first_row_in_archive_order":{"model":"SCAM","paper":"/paper/scam-transferring-humans-between-images-with","metrics":{"PSNR":"20"},"code_links":[{"title":"nicolas-dufour/SCAM","url":"https://github.com/nicolas-dufour/SCAM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-recognition-on-ade20k","task":"Scene Recognition","dataset_variant":"ADE20K","rows":1,"metrics":["Top 1 Accuracy"],"first_row_in_archive_order":{"model":"Semantic-Aware Scene Recogniton (ResNet-18)","paper":"/paper/semantic-aware-scene-recognition","metrics":{"Top 1 Accuracy":"62.55"},"code_links":[{"title":"vpulab/Semantic-Aware-Scene-Recognition","url":"https://github.com/vpulab/Semantic-Aware-Scene-Recognition"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-understanding-on-ade20k-val-1","task":"Scene Understanding","dataset_variant":"ADE20K val","rows":1,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"CPN(ResNet-101)","paper":"/paper/context-prior-for-scene-segmentation","metrics":{"Mean IoU":"46.3"},"code_links":[{"title":"ycszen/ContextPrior","url":"https://github.com/ycszen/ContextPrior"},{"title":"AndPuQing/ContextPrior_Paddle","url":"https://github.com/AndPuQing/ContextPrior_Paddle"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semi-supervised-instance-segmentation-on-1","task":"Semi-Supervised Instance Segmentation","dataset_variant":"ADE20K","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"CAST","paper":"/paper/cast-contrastive-adaptation-and-distillation","metrics":{"AP":"16.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-semantic-segmentation-on-20","task":"Weakly-Supervised Semantic Segmentation","dataset_variant":"ADE20K val","rows":1,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"DHR (Swin-L, Mask2Former)","paper":"/paper/dhr-dual-features-driven-hierarchical","metrics":{"mIoU":"32.9"},"code_links":[{"title":"shjo-april/DHR","url":"https://github.com/shjo-april/DHR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-semantic-segmentation-on-ade20k-847","task":"Zero-Shot Semantic Segmentation","dataset_variant":"ADE20K-847","rows":1,"metrics":["unseen mIoU"],"first_row_in_archive_order":{"model":"MAFT","paper":null,"metrics":{"unseen mIoU":"8.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-semantic-segmentation-on-17","task":"Open-Vocabulary Semantic Segmentation","dataset_variant":"ADE20K-150","rows":0,"metrics":["mIoU"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-semantic-segmentation-on-18","task":"Open Vocabulary Semantic Segmentation","dataset_variant":"ADE20K-150","rows":0,"metrics":["mIoU"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-segmentation-on-ade20k-150-1","task":"Semantic Segmentation","dataset_variant":"ADE20K-150","rows":0,"metrics":["mIOU"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/textregion-text-aligned-region-tokens-from","title":"TextRegion: Text-Aligned Region Tokens from Frozen Image-Text Models","date":"2025-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cast-contrastive-adaptation-and-distillation","title":"CAST: Contrastive Adaptation and Distillation for Semi-Supervised Instance Segmentation","date":"2025-05-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/the-missing-point-in-vision-transformers-for","title":"The Missing Point in Vision Transformers for Universal Image Segmentation","date":"2025-05-26","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/your-vit-is-secretly-an-image-segmentation-1","title":"Your ViT is Secretly an Image Segmentation Model","date":"2025-03-24","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/maskclip-a-mask-based-clip-fine-tuning","title":"MaskCLIP++: A Mask-Based CLIP Fine-tuning Framework for Open-Vocabulary Image Segmentation","date":"2024-12-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/mask-adapter-the-devil-is-in-the-masks-for","title":"Mask-Adapter: The Devil is in the Masks for Open-Vocabulary Segmentation","date":"2024-12-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/cosmos-cross-modality-self-distillation-for","title":"COSMOS: Cross-Modality Self-Distillation for Vision Language Pre-training","date":"2024-12-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/corrclip-reconstructing-correlations-in-clip","title":"CorrCLIP: Reconstructing Correlations in CLIP with Off-the-Shelf Foundation Models for Open-Vocabulary Semantic Segmentation","date":"2024-11-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/harnessing-vision-foundation-models-for-high","title":"Harnessing Vision Foundation Models for High-Performance, Training-Free Open Vocabulary Segmentation","date":"2024-11-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-decoders-for-transformer-based","title":"Rethinking Decoders for Transformer-based Semantic Segmentation: A Compression Perspective","date":"2024-11-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/unimatch-v2-pushing-the-limit-of-semi","title":"UniMatch V2: Pushing the Limit of Semi-Supervised Semantic Segmentation","date":"2024-10-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/debiformer-vision-transformer-with-deformable","title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","date":"2024-10-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/proxyclip-proxy-attention-improves-clip-for","title":"ProxyCLIP: Proxy Attention Improves CLIP for Open-Vocabulary Segmentation","date":"2024-08-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/in-defense-of-lazy-visual-grounding-for-open","title":"In Defense of Lazy Visual Grounding for Open-Vocabulary Semantic Segmentation","date":"2024-08-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/collaborative-vision-text-representation","title":"Collaborative Vision-Text Representation Optimizing for Open-Vocabulary Segmentation","date":"2024-08-01","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/colormae-exploring-data-independent-masking","title":"ColorMAE: Exploring data-independent masking strategies in Masked AutoEncoders","date":"2024-07-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mitigating-background-shift-in-class","title":"Mitigating Background Shift in Class-Incremental Semantic Segmentation","date":"2024-07-16","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with-4","title":"Open-Vocabulary Semantic Segmentation with Image Embedding Balancing","date":"2024-06-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":3,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/separating-the-chirp-from-the-chat-self","title":"Separating the \"Chirp\" from the \"Chat\": Self-supervised Visual Grounding of Sound and Language","date":"2024-06-09","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/parameter-inverted-image-pyramid-networks","title":"Parameter-Inverted Image Pyramid Networks","date":"2024-06-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/opendas-domain-adaptation-for-open-vocabulary","title":"OpenDAS: Open-Vocabulary Domain Adaptation for 2D and 3D Segmentation","date":"2024-05-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ttd-text-tag-self-distillation-enhancing","title":"TTD: Text-Tag Self-Distillation Enhancing Image-Text Alignment in CLIP to Alleviate Single Tag Bias","date":"2024-03-30","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dhr-dual-features-driven-hierarchical","title":"DHR: Dual Features-Driven Hierarchical Rebalancing in Inter- and Intra-Class Regions for Weakly-Supervised Semantic Segmentation","date":"2024-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":9,"samples_unverified":0,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/possam-panoptic-open-vocabulary-segment-1","title":"PosSAM: Panoptic Open-vocabulary Segment Anything","date":"2024-03-14","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vit-comer-vision-transformer-with","title":"ViT-CoMer: Vision Transformer with Convolutional Multi-scale Feature Interaction for Dense Predictions","date":"2024-03-13","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/stochastic-conditional-diffusion-models-for","title":"Stochastic Conditional Diffusion Models for Robust Semantic Image Synthesis","date":"2024-02-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sernet-former-semantic-segmentation-by","title":"SERNet-Former: Semantic Segmentation by Efficient Residual Network with Attention-Boosting Gates and Attention-Fusion Networks","date":"2024-01-28","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/umg-clip-a-unified-multi-granularity-vision","title":"UMG-CLIP: A Unified Multi-Granularity Vision Generalist for Open-World Understanding","date":"2024-01-12","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/harnessing-diffusion-models-for-visual","title":"Harnessing Diffusion Models for Visual Perception with Meta Prompts","date":"2023-12-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tagalign-improving-vision-language-alignment","title":"TagAlign: Improving Vision-Language Alignment with Multi-Tag Classification","date":"2023-12-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unlocking-pre-trained-image-backbones-for","title":"Unlocking Pre-trained Image Backbones for Semantic Image Synthesis","date":"2023-12-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/open-vocabulary-segmentation-with-semantic","title":"Open-Vocabulary Segmentation with Semantic-Assisted Calibration","date":"2023-12-07","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":5,"samples_unverified":5,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transnext-robust-foveal-visual-perception-for","title":"TransNeXt: Robust Foveal Visual Perception for Vision Transformers","date":"2023-11-28","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unireplknet-a-universal-perception-large","title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","date":"2023-11-27","rows_on_this_dataset":6,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semivl-semi-supervised-semantic-segmentation","title":"SemiVL: Semi-Supervised Semantic Segmentation with Vision-Language Guidance","date":"2023-11-27","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/sed-a-simple-encoder-decoder-for-open","title":"SED: A Simple Encoder-Decoder for Open-Vocabulary Semantic Segmentation","date":"2023-11-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/silc-improving-vision-language-pretraining","title":"SILC: Improving Vision Language Pretraining with Self-Distillation","date":"2023-10-20","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/clipself-vision-transformer-distills-itself","title":"CLIPSelf: Vision Transformer Distills Itself for Open-Vocabulary Dense Prediction","date":"2023-10-02","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-mask-aware-clip-representations-for","title":"Learning Mask-aware CLIP Representations for Zero-Shot Segmentation","date":"2023-09-30","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-image-alignment-for-diffusion-based","title":"Text-image Alignment for Diffusion-based Perception","date":"2023-09-29","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/dat-spatially-dynamic-vision-transformer-with","title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","date":"2023-09-04","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/convolutions-die-hard-open-vocabulary-1","title":"Convolutions Die Hard: Open-Vocabulary Segmentation with Single Frozen Convolutional CLIP","date":"2023-08-04","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/conditional-boundary-loss-for-semantic","title":"Conditional Boundary Loss for Semantic Segmentation","date":"2023-07-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/segvitv2-exploring-efficient-and-continual","title":"SegViTv2: Exploring Efficient and Continual Semantic Segmentation with Plain Vision Transformers","date":"2023-06-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/wavelet-based-unsupervised-label-to-image-1","title":"Wavelet-based Unsupervised Label-to-Image Translation","date":"2023-05-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/imagebind-one-embedding-space-to-bind-them","title":"ImageBind: One Embedding Space To Bind Them All","date":"2023-05-09","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":24,"samples_unverified":10,"pointer_only_for_licence":32,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/understanding-gaussian-attention-bias-of","title":"Understanding Gaussian Attention Bias of Vision Transformers Using Effective Receptive Fields","date":"2023-05-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dinov2-learning-robust-visual-features","title":"DINOv2: Learning Robust Visual Features without Supervision","date":"2023-04-14","rows_on_this_dataset":1,"code_links":26,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":46,"samples_ran":21,"samples_unverified":25,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/freeseg-unified-universal-and-open-vocabulary","title":"FreeSeg: Unified, Universal and Open-Vocabulary Image Segmentation","date":"2023-03-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ddp-diffusion-model-for-dense-visual","title":"DDP: Diffusion Model for Dense Visual Prediction","date":"2023-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fastvit-a-fast-hybrid-vision-transformer","title":"FastViT: A Fast Hybrid Vision Transformer using Structural Reparameterization","date":"2023-03-24","rows_on_this_dataset":4,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cat-seg-cost-aggregation-for-open-vocabulary","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","date":"2023-03-21","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biformer-vision-transformer-with-bi-level","title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","date":"2023-03-15","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-simple-framework-for-open-vocabulary","title":"A Simple Framework for Open-Vocabulary Segmentation and Detection","date":"2023-03-14","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/open-vocabulary-panoptic-segmentation-with-1","title":"Open-Vocabulary Panoptic Segmentation with Text-to-Image Diffusion Models","date":"2023-03-08","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/side-adapter-network-for-open-vocabulary","title":"Side Adapter Network for Open-Vocabulary Semantic Segmentation","date":"2023-02-23","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/convnext-v2-co-designing-and-scaling-convnets","title":"ConvNeXt V2: Co-designing and Scaling ConvNets with Masked Autoencoders","date":"2023-01-02","rows_on_this_dataset":9,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":10,"samples_unverified":4,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masqclip-for-open-vocabulary-universal-image","title":"MasQCLIP for Open-Vocabulary Universal Image Segmentation","date":"2023-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/label-guided-knowledge-distillation-for","title":"Label-Guided Knowledge Distillation for Continual Semantic Segmentation on 2D Images and 3D Point Clouds","date":"2023-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/representation-separation-for-semantic","title":"Representation Separation for Semantic Segmentation with Vision Transformers","date":"2022-12-28","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/reversible-column-networks","title":"Reversible Column Networks","date":"2022-12-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/generalized-decoding-for-pixel-image-and","title":"Generalized Decoding for Pixel, Image, and Language","date":"2022-12-21","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with-2","title":"Open Vocabulary Semantic Segmentation with Patch Aligned Contrastive Learning","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-to-generate-text-grounded-mask-for","title":"Learning to Generate Text-grounded Mask for Open-world Semantic Segmentation from Only Image-Text Pairs","date":"2022-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/comformer-continual-learning-in-semantic-and","title":"CoMFormer: Continual Learning in Semantic and Panoptic Segmentation","date":"2022-11-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/towards-all-in-one-pre-training-via","title":"Towards All-in-one Pre-training via Maximizing Multi-modal Mutual Information","date":"2022-11-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/oneformer-one-transformer-to-rule-universal","title":"OneFormer: One Transformer to Rule Universal Image Segmentation","date":"2022-11-10","rows_on_this_dataset":17,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internimage-exploring-large-scale-vision","title":"InternImage: Exploring Large-Scale Vision Foundation Models with Deformable Convolutions","date":"2022-11-10","rows_on_this_dataset":7,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-multi-order-gated-aggregation","title":"MogaNet: Multi-order Gated Aggregation Network","date":"2022-11-07","rows_on_this_dataset":5,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":12,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/could-giant-pretrained-image-models-extract","title":"Could Giant Pretrained Image Models Extract Universal Representations?","date":"2022-11-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/towards-sustainable-self-supervised-learning","title":"Towards Sustainable Self-supervised Learning","date":"2022-10-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/perceptual-grouping-in-vision-language-models","title":"Perceptual Grouping in Contrastive Vision-Language Models","date":"2022-10-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/attribution-aware-weight-transfer-a-warm","title":"Attribution-aware Weight Transfer: A Warm-Start Initialization for Class-Incremental Semantic Segmentation","date":"2022-10-13","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/segvit-semantic-segmentation-with-plain","title":"SegViT: Semantic Segmentation with Plain Vision Transformers","date":"2022-10-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/scam-transferring-humans-between-images-with","title":"SCAM! Transferring humans between images with Semantic Cross Attention Modulation","date":"2022-10-10","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with","title":"Open-Vocabulary Semantic Segmentation with Mask-adapted CLIP","date":"2022-10-09","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sequential-ensembling-for-semantic","title":"Sequential Ensembling for Semantic Segmentation","date":"2022-10-08","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/dual-pyramid-generative-adversarial-networks","title":"Dual Pyramid Generative Adversarial Networks for Semantic Image Synthesis","date":"2022-10-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/moat-alternating-mobile-convolution-and","title":"MOAT: Alternating Mobile Convolution and Attention Brings Strong Vision Models","date":"2022-10-04","rows_on_this_dataset":7,"code_links":2,"syntology":null},{"paper":"/paper/contrastive-audio-visual-masked-autoencoder","title":"Contrastive Audio-Visual Masked Autoencoder","date":"2022-10-02","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dilated-neighborhood-attention-transformer","title":"Dilated Neighborhood Attention Transformer","date":"2022-09-29","rows_on_this_dataset":10,"code_links":7,"syntology":null},{"paper":"/paper/generalized-parametric-contrastive-learning","title":"Generalized Parametric Contrastive Learning","date":"2022-09-26","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/exploring-target-representations-for-masked","title":"Exploring Target Representations for Masked Autoencoders","date":"2022-09-08","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gswin-gated-mlp-vision-model-with","title":"gSwin: Gated MLP Vision Model with Hierarchical Structure of Shifted Window","date":"2022-08-24","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/image-as-a-foreign-language-beit-pretraining","title":"Image as a Foreign Language: BEiT Pretraining for All Vision and Vision-Language Tasks","date":"2022-08-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/revisiting-weak-to-strong-consistency-in-semi","title":"Revisiting Weak-to-Strong Consistency in Semi-Supervised Semantic Segmentation","date":"2022-08-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-panoptic-segmentation-with","title":"Open-Vocabulary Universal Image Segmentation with MaskCLIP","date":"2022-08-18","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hornet-efficient-high-order-spatial","title":"HorNet: Efficient High-Order Spatial Interactions with Recursive Gated Convolutions","date":"2022-07-28","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/k-means-mask-transformer","title":"kMaX-DeepLab: k-means Mask Transformer","date":"2022-07-08","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/global-context-vision-transformers","title":"Global Context Vision Transformers","date":"2022-06-20","rows_on_this_dataset":3,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":36,"samples_ran":17,"samples_unverified":19,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reco-retrieve-and-co-segment-for-zero-shot-1","title":"ReCo: Retrieve and Co-segment for Zero-shot Transfer","date":"2022-06-14","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/mask-dino-towards-a-unified-transformer-based-1","title":"Mask DINO: Towards A Unified Transformer-based Framework for Object Detection and Segmentation","date":"2022-06-06","rows_on_this_dataset":2,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":11,"samples_unverified":2,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficientvit-enhanced-linear-attention-for","title":"EfficientViT: Multi-Scale Linear Attention for High-Resolution Dense Prediction","date":"2022-05-29","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contrastive-learning-rivals-masked-image","title":"Contrastive Learning Rivals Masked Image Modeling in Fine-tuning via Feature Distillation","date":"2022-05-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/architecture-agnostic-masked-image-modeling","title":"Architecture-Agnostic Masked Image Modeling -- From ViT back to CNN","date":"2022-05-27","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/mixmim-mixed-and-masked-image-modeling-for","title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","date":"2022-05-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-adapter-for-dense","title":"Vision Transformer Adapter for Dense Predictions","date":"2022-05-17","rows_on_this_dataset":5,"code_links":2,"syntology":null},{"paper":"/paper/neighborhood-attention-transformer","title":"Neighborhood Attention Transformer","date":"2022-04-14","rows_on_this_dataset":4,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deit-iii-revenge-of-the-vit","title":"DeiT III: Revenge of the ViT","date":"2022-04-14","rows_on_this_dataset":2,"code_links":12,"syntology":null},{"paper":"/paper/davit-dual-attention-vision-transformers","title":"DaViT: Dual Attention Vision Transformers","date":"2022-04-07","rows_on_this_dataset":4,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":8,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/region-rebalance-for-long-tailed-semantic","title":"Region Rebalance for Long-Tailed Semantic Segmentation","date":"2022-04-05","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/multimae-multi-modal-multi-task-masked","title":"MultiMAE: Multi-modal Multi-task Masked Autoencoders","date":"2022-04-04","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/dynamic-focus-aware-positional-queries-for","title":"Dynamic Focus-aware Positional Queries for Semantic Segmentation","date":"2022-04-04","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/focal-modulation-networks","title":"Focal Modulation Networks","date":"2022-03-22","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sats-self-attention-transfer-for-continual","title":"SATS: Self-Attention Transfer for Continual Semantic Segmentation","date":"2022-03-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/activemlp-an-mlp-like-architecture-with","title":"Active Token Mixer","date":"2022-03-11","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/representation-compensation-networks-for","title":"Representation Compensation Networks for Continual Semantic Segmentation","date":"2022-03-10","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/groupvit-semantic-segmentation-emerges-from","title":"GroupViT: Semantic Segmentation Emerges from Text Supervision","date":"2022-02-22","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/visual-attention-network","title":"Visual Attention Network","date":"2022-02-20","rows_on_this_dataset":6,"code_links":21,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-autoencoder-for-self-supervised","title":"Context Autoencoder for Self-Supervised Representation Learning","date":"2022-02-07","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/when-shift-operation-meets-vision-transformer","title":"When Shift Operation Meets Vision Transformer: An Extremely Simple Alternative to Attention Mechanism","date":"2022-01-26","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-convnet-for-the-2020s","title":"A ConvNet for the 2020s","date":"2022-01-10","rows_on_this_dataset":6,"code_links":54,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":80,"samples_ran":54,"samples_unverified":26,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-transformer-with-deformable-attention","title":"Vision Transformer with Deformable Attention","date":"2022-01-03","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/2112-14757","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","date":"2021-12-29","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/augmenting-convolutional-networks-with","title":"Augmenting Convolutional networks with attention-based aggregation","date":"2021-12-27","rows_on_this_dataset":8,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semask-semantically-masked-transformers-for-1","title":"SeMask: Semantically Masked Transformers for Semantic Segmentation","date":"2021-12-23","rows_on_this_dataset":15,"code_links":1,"syntology":null},{"paper":"/paper/elsa-enhanced-local-self-attention-for-vision","title":"ELSA: Enhanced Local Self-Attention for Vision Transformer","date":"2021-12-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masked-attention-mask-transformer-for","title":"Masked-attention Mask Transformer for Universal Image Segmentation","date":"2021-12-02","rows_on_this_dataset":15,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/denseclip-extract-free-dense-labels-from-clip","title":"Extract Free Dense Labels from CLIP","date":"2021-12-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/efficient-self-ensemble-framework-for-1","title":"Efficient Self-Ensemble for Semantic Segmentation","date":"2021-11-26","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/metaformer-is-actually-what-you-need-for","title":"MetaFormer Is Actually What You Need for Vision","date":"2021-11-22","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fbnetv5-neural-architecture-search-for","title":"FBNetV5: Neural Architecture Search for Multiple Tasks in One Run","date":"2021-11-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/swin-transformer-v2-scaling-up-capacity-and","title":"Swin Transformer V2: Scaling Up Capacity and Resolution","date":"2021-11-18","rows_on_this_dataset":2,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":30,"samples_ran":3,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ibot-image-bert-pre-training-with-online","title":"iBOT: Image BERT Pre-Training with Online Tokenizer","date":"2021-11-15","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","rows_on_this_dataset":2,"code_links":58,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":137,"samples_ran":71,"samples_unverified":66,"pointer_only_for_licence":73,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hrvit-multi-scale-high-resolution-vision","title":"Multi-Scale High-Resolution Vision Transformer for Semantic Segmentation","date":"2021-11-01","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/semi-supervised-semantic-segmentation-via-2","title":"Semi-Supervised Semantic Segmentation via Adaptive Equalization Learning","date":"2021-10-11","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/usis-unsupervised-semantic-image-synthesis","title":"USIS: Unsupervised Semantic Image Synthesis","date":"2021-09-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/condnet-conditional-classifier-for-scene","title":"CondNet: Conditional Classifier for Scene Segmentation","date":"2021-09-21","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/is-attention-better-than-matrix-decomposition-1","title":"Is Attention Better Than Matrix Decomposition?","date":"2021-09-09","rows_on_this_dataset":8,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/convmlp-hierarchical-convolutional-mlps-for","title":"ConvMLP: Hierarchical Convolutional MLPs for Vision","date":"2021-09-09","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fapn-feature-aligned-pyramid-network-for","title":"FaPN: Feature-aligned Pyramid Network for Dense Image Prediction","date":"2021-08-16","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/crossformer-a-versatile-vision-transformer","title":"CrossFormer: A Versatile Vision Transformer Hinging on Cross-scale Attention","date":"2021-07-31","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/per-pixel-classification-is-not-all-you-need","title":"Per-Pixel Classification is Not All You Need for Semantic Segmentation","date":"2021-07-13","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/focal-self-attention-for-local-global","title":"Focal Self-attention for Local-Global Interactions in Vision Transformers","date":"2021-07-01","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cswin-transformer-a-general-vision","title":"CSWin Transformer: A General Vision Transformer Backbone with Cross-Shaped Windows","date":"2021-07-01","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/k-net-towards-unified-image-segmentation","title":"K-Net: Towards Unified Image Segmentation","date":"2021-06-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/volo-vision-outlooker-for-visual-recognition","title":"VOLO: Vision Outlooker for Visual Recognition","date":"2021-06-24","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ssul-semantic-segmentation-with-unknown-label","title":"SSUL: Semantic Segmentation with Unknown Label for Exemplar-based Class-Incremental Learning","date":"2021-06-22","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":6,"samples_unverified":6,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/xcit-cross-covariance-image-transformers","title":"XCiT: Cross-Covariance Image Transformers","date":"2021-06-17","rows_on_this_dataset":6,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":3,"samples_unverified":11,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/beit-bert-pre-training-of-image-transformers","title":"BEiT: BERT Pre-Training of Image Transformers","date":"2021-06-15","rows_on_this_dataset":2,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/shuffle-transformer-rethinking-spatial","title":"Shuffle Transformer: Rethinking Spatial Shuffle for Vision Transformer","date":"2021-06-07","rows_on_this_dataset":5,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/segformer-simple-and-efficient-design-for","title":"SegFormer: Simple and Efficient Design for Semantic Segmentation with Transformers","date":"2021-05-31","rows_on_this_dataset":4,"code_links":28,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":86,"samples_ran":48,"samples_unverified":38,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/segmenter-transformer-for-semantic","title":"Segmenter: Transformer for Semantic Segmentation","date":"2021-05-12","rows_on_this_dataset":6,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-learning-with-swin","title":"Self-Supervised Learning with Swin Transformers","date":"2021-05-10","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/beyond-self-attention-external-attention","title":"Beyond Self-attention: External Attention using Two Linear Layers for Visual Tasks","date":"2021-05-05","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/cassod-net-cascaded-and-separable-structures","title":"CASSOD-Net: Cascaded and Separable Structures of Dilated Convolution for Embedded Vision Systems and Applications","date":"2021-04-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/twins-revisiting-spatial-attention-design-in","title":"Twins: Revisiting the Design of Spatial Attention in Vision Transformers","date":"2021-04-28","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improve-vision-transformers-training-by","title":"Vision Transformers with Patch Diversification","date":"2021-04-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/token-labeling-training-a-85-5-top-1-accuracy","title":"All Tokens Matter: Token Labeling for Training Better Vision Transformers","date":"2021-04-22","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ctnet-context-based-tandem-network-for","title":"CTNet: Context-based Tandem Network for Semantic Segmentation","date":"2021-04-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","rows_on_this_dataset":4,"code_links":80,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":207,"samples_ran":108,"samples_unverified":99,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-transformers-for-dense-prediction","title":"Vision Transformers for Dense Prediction","date":"2021-03-24","rows_on_this_dataset":2,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":116,"samples_ran":54,"samples_unverified":62,"pointer_only_for_licence":15,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diverse-semantic-image-synthesis-via","title":"Diverse Semantic Image Synthesis via Probability Distribution Modeling","date":"2021-03-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-semantic-segmentation-from-a","title":"Rethinking Semantic Segmentation from a Sequence-to-Sequence Perspective with Transformers","date":"2020-12-31","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/taming-transformers-for-high-resolution-image","title":"Taming Transformers for High-Resolution Image Synthesis","date":"2020-12-17","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/you-only-need-adversarial-supervision-for-1","title":"You Only Need Adversarial Supervision for Semantic Image Synthesis","date":"2020-12-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improving-augmentation-and-evaluation-schemes","title":"Improving Augmentation and Evaluation Schemes for Semantic Image Synthesis","date":"2020-11-25","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/plop-learning-without-forgetting-for","title":"PLOP: Learning without Forgetting for Continual Semantic Segmentation","date":"2020-11-23","rows_on_this_dataset":8,"code_links":2,"syntology":null},{"paper":"/paper/scene-segmentation-with-dual-relation-aware","title":"Scene Segmentation with Dual Relation-aware Attention Network","date":"2020-08-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pyramidal-convolution-rethinking","title":"Pyramidal Convolution: Rethinking Convolutional Neural Networks for Visual Recognition","date":"2020-06-20","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/disentangled-non-local-neural-networks","title":"Disentangled Non-Local Neural Networks","date":"2020-06-11","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/resnest-split-attention-networks","title":"ResNeSt: Split-Attention Networks","date":"2020-04-19","rows_on_this_dataset":6,"code_links":36,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":8,"samples_unverified":40,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-domain-correspondence-learning-for","title":"Cross-domain Correspondence Learning for Exemplar-based Image Translation","date":"2020-04-12","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sesame-semantic-editing-of-scenes-by-adding","title":"SESAME: Semantic Editing of Scenes by Adding, Manipulating or Erasing Objects","date":"2020-04-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/context-prior-for-scene-segmentation","title":"Context Prior for Scene Segmentation","date":"2020-04-03","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/muxconv-information-multiplexing-in","title":"MUXConv: Information Multiplexing in Convolutional Neural Networks","date":"2020-03-31","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/2003-13328","title":"Strip Pooling: Rethinking Spatial Pooling for Scene Parsing","date":"2020-03-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dcnas-densely-connected-neural-architecture","title":"DCNAS: Densely Connected Neural Architecture Search for Semantic Image Segmentation","date":"2020-03-26","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/location-aware-upsampling-for-semantic","title":"Location-aware Upsampling for Semantic Segmentation","date":"2019-11-13","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/adaptive-context-network-for-scene-parsing-1","title":"Adaptive Context Network for Scene Parsing","date":"2019-11-05","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/learning-to-predict-layout-to-image","title":"Learning to Predict Layout-to-image Conditional Convolutions for Semantic Image Synthesis","date":"2019-10-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/object-contextual-representations-for","title":"Segmentation Transformer: Object-Contextual Representations for Semantic Segmentation","date":"2019-09-24","rows_on_this_dataset":6,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semantic-aware-scene-recognition","title":"Semantic-Aware Scene Recognition","date":"2019-09-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/asymmetric-non-local-neural-networks-for","title":"Asymmetric Non-local Neural Networks for Semantic Segmentation","date":"2019-08-21","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/incremental-learning-techniques-for-semantic","title":"Incremental Learning Techniques for Semantic Segmentation","date":"2019-07-31","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/consistency-regularization-and-cutmix-for","title":"Semi-supervised semantic segmentation needs strong, varied perturbations","date":"2019-06-05","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/co-occurrent-features-in-semantic","title":"Co-Occurrent Features in Semantic Segmentation","date":"2019-06-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/high-resolution-representations-for-labeling","title":"High-Resolution Representations for Labeling Pixels and Regions","date":"2019-04-09","rows_on_this_dataset":2,"code_links":39,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":3,"samples_unverified":15,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fastfcn-rethinking-dilated-convolution-in-the","title":"FastFCN: Rethinking Dilated Convolution in the Backbone for Semantic Segmentation","date":"2019-03-28","rows_on_this_dataset":1,"code_links":12,"syntology":null},{"paper":"/paper/semantic-image-synthesis-with-spatially","title":"Semantic Image Synthesis with Spatially-Adaptive Normalization","date":"2019-03-18","rows_on_this_dataset":2,"code_links":24,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":5,"samples_unverified":4,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/auto-deeplab-hierarchical-neural-architecture","title":"Auto-DeepLab: Hierarchical Neural Architecture Search for Semantic Image Segmentation","date":"2019-01-10","rows_on_this_dataset":2,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/symbolic-graph-reasoning-meets-convolutions","title":"Symbolic Graph Reasoning Meets Convolutions","date":"2018-12-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/psanet-point-wise-spatial-attention-network","title":"PSANet: Point-wise Spatial Attention Network for Scene Parsing","date":"2018-09-01","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/unified-perceptual-parsing-for-scene","title":"Unified Perceptual Parsing for Scene Understanding","date":"2018-07-26","rows_on_this_dataset":2,"code_links":25,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":28,"samples_ran":9,"samples_unverified":19,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semi-parametric-image-synthesis","title":"Semi-parametric Image Synthesis","date":"2018-04-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/jointly-discovering-visual-objects-and-spoken","title":"Jointly Discovering Visual Objects and Spoken Words from Raw Sensory Input","date":"2018-04-04","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/context-encoding-for-semantic-segmentation","title":"Context Encoding for Semantic Segmentation","date":"2018-03-23","rows_on_this_dataset":2,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dynamic-structured-semantic-propagation","title":"Dynamic-structured Semantic Propagation Network","date":"2018-03-16","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/high-resolution-image-synthesis-and-semantic","title":"High-Resolution Image Synthesis and Semantic Manipulation with Conditional GANs","date":"2017-11-30","rows_on_this_dataset":2,"code_links":21,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/photographic-image-synthesis-with-cascaded","title":"Photographic Image Synthesis with Cascaded Refinement Networks","date":"2017-07-28","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/pyramid-scene-parsing-network","title":"Pyramid Scene Parsing Network","date":"2016-12-04","rows_on_this_dataset":5,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":29,"samples_ran":7,"samples_unverified":22,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/refinenet-multi-path-refinement-networks-for","title":"RefineNet: Multi-Path Refinement Networks for High-Resolution Semantic Segmentation","date":"2016-11-20","rows_on_this_dataset":3,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-scale-context-aggregation-by-dilated","title":"Multi-Scale Context Aggregation by Dilated Convolutions","date":"2015-11-23","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/segnet-a-deep-convolutional-encoder-decoder","title":"SegNet: A Deep Convolutional Encoder-Decoder Architecture for Image Segmentation","date":"2015-11-02","rows_on_this_dataset":1,"code_links":74,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":44,"samples_ran":9,"samples_unverified":35,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fully-convolutional-networks-for-semantic-1","title":"Fully Convolutional Networks for Semantic Segmentation","date":"2014-11-14","rows_on_this_dataset":1,"code_links":51,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":102,"samples_harvested":1462,"samples_ran":672,"samples_unverified":790,"pointer_only_for_licence":418,"papers_with_no_sample_that_ran":15,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}