{"url":"/dataset/pascal-voc","name":"PASCAL VOC","full_name":"PASCAL Visual Object Classes Challenge","description_markdown":"The PASCAL Visual Object Classes (VOC) 2012 dataset contains 20 object categories including vehicles, household, animals, and other: aeroplane, bicycle, boat, bus, car, motorbike, train, bottle, chair, dining table, potted plant, sofa, TV/monitor, bird, cat, cow, dog, horse, sheep, and person. Each image in this dataset has pixel-level segmentation annotations, bounding box annotations, and object class annotations. This dataset has been widely used as a benchmark for object detection, semantic segmentation, and classification tasks. The **PASCAL VOC** dataset is split into three subsets: 1,464 images for training, 1,449 images for validation and a private testing set.\r\n\r\nSource: [Self-supervised Visual Feature Learning with Deep Neural Networks: A Survey](https://arxiv.org/abs/1902.06162)\r\nImage Source: [http://host.robots.ox.ac.uk/pascal/VOC/voc2012/examples/images/sheep_06.jpg](http://host.robots.ox.ac.uk/pascal/VOC/voc2012/examples/images/sheep_06.jpg)","description_withheld":null,"homepage":"http://host.robots.ox.ac.uk/pascal/VOC/","introduced_date":"2010-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/location-aware-single-image-reflection","title":"Location-aware Single Image Reflection Removal","first_author":"Zheng Dong","url":null},"license":{"name":"Custom","url":"http://host.robots.ox.ac.uk/pascal/VOC/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Semantic Segmentation","url":"/task/semantic-segmentation","datasets_with_task":"/datasets/task/semantic-segmentation"},{"name":"Node Classification","url":"/task/node-classification","datasets_with_task":"/datasets/task/node-classification"},{"name":"3D Face Animation","url":"/task/3d-face-animation","datasets_with_task":"/datasets/task/3d-face-animation"},{"name":"Object Counting","url":"/task/object-counting","datasets_with_task":"/datasets/task/object-counting"},{"name":"Interactive Segmentation","url":"/task/interactive-segmentation","datasets_with_task":"/datasets/task/interactive-segmentation"},{"name":"Image Segmentation","url":"/task/image-segmentation","datasets_with_task":"/datasets/task/image-segmentation"},{"name":"Open Vocabulary Semantic Segmentation","url":"/task/open-vocabulary-semantic-segmentation","datasets_with_task":"/datasets/task/open-vocabulary-semantic-segmentation"},{"name":"Unsupervised Semantic Segmentation with Language-image Pre-training","url":"/task/unsupervised-semantic-segmentation-with","datasets_with_task":"/datasets/task/unsupervised-semantic-segmentation-with"},{"name":"Knowledge Distillation","url":"/task/knowledge-distillation","datasets_with_task":"/datasets/task/knowledge-distillation"},{"name":"Graph Matching","url":"/task/graph-matching","datasets_with_task":"/datasets/task/graph-matching"},{"name":"Single-object discovery","url":"/task/single-object-discovery","datasets_with_task":"/datasets/task/single-object-discovery"},{"name":"Single-object colocalization","url":"/task/single-object-colocalization","datasets_with_task":"/datasets/task/single-object-colocalization"},{"name":"Zero-Shot Semantic Segmentation","url":"/task/zero-shot-semantic-segmentation","datasets_with_task":"/datasets/task/zero-shot-semantic-segmentation"},{"name":"Multi-object discovery","url":"/task/multi-object-discovery","datasets_with_task":"/datasets/task/multi-object-discovery"},{"name":"Talking Face Generation","url":"/task/talking-face-generation","datasets_with_task":"/datasets/task/talking-face-generation"},{"name":"Multi-object colocalization","url":"/task/multi-object-colocalization","datasets_with_task":"/datasets/task/multi-object-colocalization"}],"languages":[],"variants":["Pascal VOC 2007 count-test","VOCASET","VOC_all","VOC_6x2","PascalVOC-SP","PascalVOC-59","PascalVOC-459","PascalVOC-20b","PascalVOC-20","PASCAL VOC 10%","PASCAL VOC"],"data_loaders":[{"repo":"https://github.com/facebookresearch/detectron2","url":"https://detectron2.readthedocs.io/en/latest/tutorials/builtin_datasets.html#expected-dataset-structure-for-pascal-voc","frameworks":["pytorch"]},{"repo":"https://github.com/open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection/blob/master/docs/1_exist_data_model.md","frameworks":["pytorch"]},{"repo":"https://github.com/pytorch/vision","url":"https://pytorch.org/vision/stable/generated/torchvision.datasets.VOCDetection.html","frameworks":["pytorch"]},{"repo":"https://github.com/open-mmlab/mmsegmentation","url":"https://github.com/open-mmlab/mmsegmentation/blob/master/docs/dataset_prepare.md","frameworks":["pytorch"]},{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/pascal-voc-2012-dataset","frameworks":[]}],"num_papers_in_archive":198,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/graph-matching-on-pascal-voc","task":"Graph Matching","dataset_variant":"PASCAL VOC","rows":31,"metrics":["F1 score","matching accuracy"],"first_row_in_archive_order":{"model":"URL","paper":"/paper/universe-points-representation-learning-for","metrics":{"F1 score":"0.717±0.005","matching accuracy":"0.818"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/node-classification-on-pascalvoc-sp-1","task":"Node Classification","dataset_variant":"PascalVOC-SP","rows":21,"metrics":["macro F1"],"first_row_in_archive_order":{"model":"NeuralWalker","paper":"/paper/learning-long-range-dependencies-on-graphs","metrics":{"macro F1":"0.4912 ± 0.0042"},"code_links":[{"title":"borgwardtlab/neuralwalker","url":"https://github.com/borgwardtlab/neuralwalker"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-semantic-segmentation-on-5","task":"Open Vocabulary Semantic Segmentation","dataset_variant":"PascalVOC-20","rows":20,"metrics":["mIoU","hIoU"],"first_row_in_archive_order":{"model":"UMG-CLIP-L/14","paper":"/paper/umg-clip-a-unified-multi-granularity-vision","metrics":{"mIoU":"97.9"},"code_links":[{"title":"lygsbw/umg-clip","url":"https://github.com/lygsbw/umg-clip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-semantic-segmentation-on-pascal-voc","task":"Zero-Shot Semantic Segmentation","dataset_variant":"PASCAL VOC","rows":13,"metrics":["Transductive Setting hIoU","Inductive Setting hIoU"],"first_row_in_archive_order":{"model":"OTSeg+","paper":"/paper/otseg-multi-prompt-sinkhorn-attention-for","metrics":{"Inductive Setting hIoU":"87.4","Transductive Setting hIoU":"94.4"},"code_links":[{"title":"cubeyoung/OTSeg","url":"https://github.com/cubeyoung/OTSeg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-semantic-segmentation-with-11","task":"Unsupervised Semantic Segmentation with Language-image Pre-training","dataset_variant":"PASCAL VOC","rows":10,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"CorrCLIP","paper":"/paper/corrclip-reconstructing-correlations-in-clip","metrics":{"mIoU":"76.7"},"code_links":[{"title":"zdk258/CorrCLIP","url":"https://github.com/zdk258/CorrCLIP"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-semantic-segmentation-with-7","task":"Unsupervised Semantic Segmentation with Language-image Pre-training","dataset_variant":"PascalVOC-20","rows":10,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"CorrCLIP","paper":"/paper/corrclip-reconstructing-correlations-in-clip","metrics":{"mIoU":"91.8"},"code_links":[{"title":"zdk258/CorrCLIP","url":"https://github.com/zdk258/CorrCLIP"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-vocabulary-semantic-segmentation-on-9","task":"Open Vocabulary Semantic Segmentation","dataset_variant":"PascalVOC-20b","rows":4,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"UMG-CLIP-E/14","paper":"/paper/umg-clip-a-unified-multi-granularity-vision","metrics":{"mIoU":"85.4"},"code_links":[{"title":"lygsbw/umg-clip","url":"https://github.com/lygsbw/umg-clip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-segmentation-on-pascal-voc","task":"Image Segmentation","dataset_variant":"PASCAL VOC","rows":3,"metrics":["mIoU","mAP0.5"],"first_row_in_archive_order":{"model":"OneNete,4-C","paper":"/paper/onenet-a-channel-wise-1d-convolutional-u-net","metrics":{"mIoU":"63.6"},"code_links":[{"title":"shbyun080/onenet","url":"https://github.com/shbyun080/onenet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-counting-on-pascal-voc","task":"Object Counting","dataset_variant":"PASCAL VOC","rows":3,"metrics":["mRMSE"],"first_row_in_archive_order":{"model":"TFOC","paper":"/paper/training-free-object-counting-with-prompts","metrics":{"mRMSE":"0.0084"},"code_links":[{"title":"shizenglin/training-free-object-counter","url":"https://github.com/shizenglin/training-free-object-counter"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/single-object-discovery-on-voc-all","task":"Single-object discovery","dataset_variant":"VOC_all","rows":3,"metrics":["CorLoc"],"first_row_in_archive_order":{"model":"Large-scale rOSD","paper":"/paper/toward-unsupervised-multi-object-discovery-in","metrics":{"CorLoc":"49.4"},"code_links":[{"title":"huyvvo/rOSD","url":"https://github.com/huyvvo/rOSD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/interactive-segmentation-on-pascal-voc","task":"Interactive Segmentation","dataset_variant":"PASCAL VOC","rows":2,"metrics":["NoC@95","NoC@90","NoC@85"],"first_row_in_archive_order":{"model":"ICL CFR-1 (ViT-H, C+L)","paper":"/paper/cfr-icl-cascade-forward-refinement-with","metrics":{"NoC@85":"1.72","NoC@90":"1.94","NoC@95":"2.45"},"code_links":[{"title":"TitorX/CFR-ICL-Interactive-Segmentation","url":"https://github.com/TitorX/CFR-ICL-Interactive-Segmentation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/knowledge-distillation-on-pascal-voc","task":"Knowledge Distillation","dataset_variant":"PASCAL VOC","rows":2,"metrics":["mAP"],"first_row_in_archive_order":{"model":"LSHFM (T: ResNet101 S: ResNet50)","paper":"/paper/in-defense-of-feature-mimicking-for-knowledge","metrics":{"mAP":"93.17"},"code_links":[{"title":"DoctorKey/LSHFM.singleclassification","url":"https://github.com/DoctorKey/LSHFM.singleclassification"},{"title":"DoctorKey/LSHFM.detection","url":"https://github.com/DoctorKey/LSHFM.detection"},{"title":"DoctorKey/LSHFM.multiclassification","url":"https://github.com/DoctorKey/LSHFM.multiclassification"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-object-discovery-on-voc-all","task":"Multi-object discovery","dataset_variant":"VOC_all","rows":2,"metrics":["Detection Rate"],"first_row_in_archive_order":{"model":"Large-scale rOSD","paper":"/paper/toward-unsupervised-multi-object-discovery-in","metrics":{"Detection Rate":"38.3"},"code_links":[{"title":"huyvvo/rOSD","url":"https://github.com/huyvvo/rOSD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-pascal-voc-10","task":"Object Detection","dataset_variant":"PASCAL VOC 10%","rows":2,"metrics":["AP","AP50","AP75"],"first_row_in_archive_order":{"model":"DETReg (MDef-DETR)","paper":"/paper/multi-modal-transformers-excel-at-class","metrics":{"AP":"58.78","AP50":"80.46","AP75":"65.65"},"code_links":[{"title":"mmaaz60/mvits_for_class_agnostic_od","url":"https://github.com/mmaaz60/mvits_for_class_agnostic_od"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/single-object-discovery-on-voc-6x2","task":"Single-object discovery","dataset_variant":"VOC_6x2","rows":2,"metrics":["CorLoc"],"first_row_in_archive_order":{"model":"rOSD","paper":"/paper/toward-unsupervised-multi-object-discovery-in","metrics":{"CorLoc":"72.5"},"code_links":[{"title":"huyvvo/rOSD","url":"https://github.com/huyvvo/rOSD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-object-colocalization-on-voc-all","task":"Multi-object colocalization","dataset_variant":"VOC_all","rows":1,"metrics":["Detection Rate"],"first_row_in_archive_order":{"model":"rOSD","paper":"/paper/toward-unsupervised-multi-object-discovery-in","metrics":{"Detection Rate":"49.4"},"code_links":[{"title":"huyvvo/rOSD","url":"https://github.com/huyvvo/rOSD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-pascal-voc","task":"Object Detection","dataset_variant":"PASCAL VOC","rows":1,"metrics":["Parameters(K)"],"first_row_in_archive_order":{"model":"TinyissimoYOLO-v8","paper":"/paper/ultra-efficient-on-device-object-detection-on","metrics":{"Parameters(K)":"839"},"code_links":[{"title":"eth-pbl/tinyissimoyolo","url":"https://github.com/eth-pbl/tinyissimoyolo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-segmentation-on-pascal-voc","task":"Semantic Segmentation","dataset_variant":"PASCAL VOC","rows":1,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"SegCLIP","paper":"/paper/segclip-patch-aggregation-with-learnable","metrics":{"mIoU":"52.6"},"code_links":[{"title":"arrowluo/segclip","url":"https://github.com/arrowluo/segclip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/textregion-text-aligned-region-tokens-from","title":"TextRegion: Text-Aligned Region Tokens from Frozen Image-Text Models","date":"2025-05-29","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/improving-the-effective-receptive-field-of","title":"Improving the Effective Receptive Field of Message-Passing Neural Networks","date":"2025-05-29","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unlocking-the-potential-of-classic-gnns-for","title":"Unlocking the Potential of Classic GNNs for Graph-level Tasks: Simple Architectures Meet Excellence","date":"2025-02-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":4,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/maskclip-a-mask-based-clip-fine-tuning","title":"MaskCLIP++: A Mask-Based CLIP Fine-tuning Framework for Open-Vocabulary Image Segmentation","date":"2024-12-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cosmos-cross-modality-self-distillation-for","title":"COSMOS: Cross-Modality Self-Distillation for Vision Language Pre-training","date":"2024-12-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/corrclip-reconstructing-correlations-in-clip","title":"CorrCLIP: Reconstructing Correlations in CLIP with Off-the-Shelf Foundation Models for Open-Vocabulary Semantic Segmentation","date":"2024-11-15","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/onenet-a-channel-wise-1d-convolutional-u-net","title":"OneNet: A Channel-Wise 1D Convolutional U-Net","date":"2024-11-14","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/harnessing-vision-foundation-models-for-high","title":"Harnessing Vision Foundation Models for High-Performance, Training-Free Open Vocabulary Segmentation","date":"2024-11-14","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/proxyclip-proxy-attention-improves-clip-for","title":"ProxyCLIP: Proxy Attention Improves CLIP for Open-Vocabulary Segmentation","date":"2024-08-09","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/in-defense-of-lazy-visual-grounding-for-open","title":"In Defense of Lazy Visual Grounding for Open-Vocabulary Semantic Segmentation","date":"2024-08-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/collaborative-vision-text-representation","title":"Collaborative Vision-Text Representation Optimizing for Open-Vocabulary Segmentation","date":"2024-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/next-level-message-passing-with-hierarchical","title":"Next Level Message-Passing with Hierarchical Support Graphs","date":"2024-06-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":4,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-semantic-segmentation-with-4","title":"Open-Vocabulary Semantic Segmentation with Image Embedding Balancing","date":"2024-06-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":3,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-long-range-dependencies-on-graphs","title":"Learning Long Range Dependencies on Graphs via Random Walks","date":"2024-06-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":13,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-latent-partial-matchings-with-gumbel","title":"Learning Latent Partial Matchings with Gumbel-IPF Networks","date":"2024-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ttd-text-tag-self-distillation-enhancing","title":"TTD: Text-Tag Self-Distillation Enhancing Image-Text Alignment in CLIP to Alleviate Single Tag Bias","date":"2024-03-30","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-retrieval-with-noisy","title":"Cross-modal Retrieval with Noisy Correspondence via Consistency Refining and Mining","date":"2024-03-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/otseg-multi-prompt-sinkhorn-attention-for","title":"OTSeg: Multi-prompt Sinkhorn Attention for Zero-Shot Semantic Segmentation","date":"2024-03-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":9,"samples_unverified":1,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/umg-clip-a-unified-multi-granularity-vision","title":"UMG-CLIP: A Unified Multi-Granularity Vision Generalist for Open-World Understanding","date":"2024-01-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/exploring-regional-clues-in-clip-for-zero","title":"Exploring Regional Clues in CLIP for Zero-Shot Semantic Segmentation","date":"2024-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tagalign-improving-vision-language-alignment","title":"TagAlign: Improving Vision-Language Alignment with Multi-Tag Classification","date":"2023-12-21","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/tagclip-a-local-to-global-framework-to","title":"TagCLIP: A Local-to-Global Framework to Enhance Open-Vocabulary Multi-Label Classification of CLIP Without Training","date":"2023-12-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-segmentation-with-semantic","title":"Open-Vocabulary Segmentation with Semantic-Assisted Calibration","date":"2023-12-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":5,"samples_unverified":5,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gmtr-graph-matching-transformers","title":"GMTR: Graph Matching Transformers","date":"2023-11-14","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ultra-efficient-on-device-object-detection-on","title":"Ultra-Efficient On-Device Object Detection on AI-Integrated Smart Glasses with TinyissimoYOLO","date":"2023-11-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/silc-improving-vision-language-pretraining","title":"SILC: Improving Vision Language Pretraining with Self-Distillation","date":"2023-10-20","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/learning-mask-aware-clip-representations-for","title":"Learning Mask-aware CLIP Representations for Zero-Shot Segmentation","date":"2023-09-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/where-did-the-gap-go-reassessing-the-long","title":"Where Did the Gap Go? Reassessing the Long-Range Graph Benchmark","date":"2023-09-01","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":5,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/convolutions-die-hard-open-vocabulary-1","title":"Convolutions Die Hard: Open-Vocabulary Segmentation with Single Frozen Convolutional CLIP","date":"2023-08-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/training-free-object-counting-with-prompts","title":"Training-free Object Counting with Prompts","date":"2023-06-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/drew-dynamically-rewired-message-passing-with","title":"DRew: Dynamically Rewired Message Passing with Delay","date":"2023-05-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/prompt-pre-training-with-twenty-thousand-1","title":"Prompt Pre-Training with Twenty-Thousand Classes for Open-Vocabulary Visual Recognition","date":"2023-04-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-semantic-segmentation-with","title":"Open-Vocabulary Semantic Segmentation with Decoupled One-Pass Network","date":"2023-04-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cat-seg-cost-aggregation-for-open-vocabulary","title":"CAT-Seg: Cost Aggregation for Open-Vocabulary Semantic Segmentation","date":"2023-03-21","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exphormer-sparse-transformers-for-graphs","title":"Exphormer: Sparse Transformers for Graphs","date":"2023-03-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cfr-icl-cascade-forward-refinement-with","title":"CFR-ICL: Cascade-Forward Refinement with Iterative Click Loss for Interactive Image Segmentation","date":"2023-03-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-panoptic-segmentation-with-1","title":"Open-Vocabulary Panoptic Segmentation with Text-to-Image Diffusion Models","date":"2023-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/on-the-connection-between-mpnn-and-graph","title":"On the Connection Between MPNN and Graph Transformer","date":"2023-01-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-learning-of-partial-graph-matching-via","title":"Deep Learning of Partial Graph Matching via Differentiable Top-K","date":"2023-01-01","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with-2","title":"Open Vocabulary Semantic Segmentation with Patch Aligned Contrastive Learning","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/graph-matching-with-bi-level-noisy","title":"Graph Matching with Bi-level Noisy Correspondence","date":"2022-12-08","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":7,"samples_unverified":0,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zegclip-towards-adapting-clip-for-zero-shot","title":"ZegCLIP: Towards Adapting CLIP for Zero-shot Semantic Segmentation","date":"2022-12-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":8,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universe-points-representation-learning-for","title":"Universe Points Representation Learning for Partial Multi-Graph Matching","date":"2022-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-to-generate-text-grounded-mask-for","title":"Learning to Generate Text-grounded Mask for Open-world Semantic Segmentation from Only Image-Text Pairs","date":"2022-12-01","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/segclip-patch-aggregation-with-learnable","title":"SegCLIP: Patch Aggregation with Learnable Centers for Open-Vocabulary Semantic Segmentation","date":"2022-11-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/open-vocabulary-semantic-segmentation-with","title":"Open-Vocabulary Semantic Segmentation with Mask-adapted CLIP","date":"2022-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/freeseg-free-mask-from-interpretable","title":"FreeSeg: Free Mask from Interpretable Contrastive Language-Image Pretraining for Semantic Segmentation","date":"2022-09-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/long-range-graph-benchmark","title":"Long Range Graph Benchmark","date":"2022-06-16","rows_on_this_dataset":7,"code_links":2,"syntology":null},{"paper":"/paper/reco-retrieve-and-co-segment-for-zero-shot-1","title":"ReCo: Retrieve and Co-segment for Zero-shot Transfer","date":"2022-06-14","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/recipe-for-a-general-powerful-scalable-graph","title":"Recipe for a General, Powerful, Scalable Graph Transformer","date":"2022-05-25","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":3,"samples_unverified":18,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-constrained-structured-spaces-with","title":"Learning Constrained Structured Spaces with Application to Multi-Graph Matching","date":"2022-05-03","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/groupvit-semantic-segmentation-emerges-from","title":"GroupViT: Semantic Segmentation Emerges from Text Supervision","date":"2022-02-22","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/graph-context-attention-networks-for-size","title":"Graph-Context Attention Networks for Size-Varied Deep Graph Matching","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/appearance-and-structure-aware-robust-deep","title":"Appearance and Structure Aware Robust Deep Visual Graph Matching: Attack, Defense and Beyond","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/2112-14757","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","date":"2021-12-29","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/decoupling-zero-shot-semantic-segmentation","title":"Decoupling Zero-Shot Semantic Segmentation","date":"2021-12-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/denseclip-extract-free-dense-labels-from-clip","title":"Extract Free Dense Labels from CLIP","date":"2021-12-02","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/multi-modal-transformers-excel-at-class","title":"Class-agnostic Object Detection with Multi-modal Transformer","date":"2021-11-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gamnet-robust-feature-matching-via-graph","title":"GAMnet: Robust Feature Matching via Graph Adversarial-Matching Network","date":"2021-10-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/edgeflow-achieving-practical-interactive","title":"EdgeFlow: Achieving Practical Interactive Segmentation with Edge-Guided Flow","date":"2021-09-20","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adaptive-edge-attention-for-graph-matching","title":"Adaptive Edge Attention for Graph Matching with Outliers","date":"2021-08-19","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/detreg-unsupervised-pretraining-with-region","title":"DETReg: Unsupervised Pretraining with Region Priors for Object Detection","date":"2021-06-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ia-gm-a-deep-bidirectional-learning-method","title":"IA-GM: A Deep Bidirectional Learning Method for Graph Matching","date":"2021-05-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-closer-look-at-self-training-for-zero-label","title":"A Closer Look at Self-training for Zero-Label Semantic Segmentation","date":"2021-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/joint-deep-multi-graph-matching-and-3d","title":"Joint Deep Multi-Graph Matching and 3D Geometry Learning from Inhomogeneous 2D Image Collections","date":"2021-03-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deep-graph-matching-under-quadratic","title":"Deep Graph Matching under Quadratic Constraint","date":"2021-03-11","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hypergraph-neural-networks-for-hypergraph","title":"Hypergraph Neural Networks for Hypergraph Matching","date":"2021-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/in-defense-of-feature-mimicking-for-knowledge","title":"Distilling Knowledge by Mimicking Features","date":"2020-11-03","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-aware-feature-generation-for-zero","title":"Context-aware Feature Generation for Zero-shot Semantic Segmentation","date":"2020-08-16","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/toward-unsupervised-multi-object-discovery-in","title":"Toward unsupervised, multi-object discovery in large-scale image collections","date":"2020-07-06","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/simple-and-deep-graph-convolutional-networks-1","title":"Simple and Deep Graph Convolutional Networks","date":"2020-07-04","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/combinatorial-learning-of-robust-deep-graph","title":"Combinatorial Learning of Robust Deep Graph Matching: an Embedding based Approach.","date":"2020-06-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-graph-matching-via-blackbox","title":"Deep Graph Matching via Blackbox Differentiation of Combinatorial Solvers","date":"2020-03-25","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-deep-graph-matching-with-channel","title":"Learning deep graph matching with channel-independent embedding and Hungarian attention","date":"2020-01-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/neural-graph-matching-network-learning","title":"Neural Graph Matching Network: Learning Lawler's Quadratic Assignment Problem with Extension to Hypergraph and Multiple-graph Matching","date":"2019-11-26","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/glmnet-graph-learning-matching-networks-for","title":"GLMNet: Graph Learning-Matching Networks for Feature Matching","date":"2019-11-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/190600817","title":"Zero-Shot Semantic Segmentation","date":"2019-06-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/semantic-projection-network-for-zero-and-few","title":"Semantic Projection Network for Zero- and Few-Label Semantic Segmentation","date":"2019-06-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unsupervised-image-matching-and-object","title":"Unsupervised Image Matching and Object Discovery as Optimization","date":"2019-04-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/object-counting-and-instance-segmentation","title":"Object Counting and Instance Segmentation with Image-level Supervision","date":"2019-03-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/deep-learning-of-graph-matching","title":"Deep Learning of Graph Matching","date":"2018-06-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/counting-everyday-objects-in-everyday-scenes","title":"Counting Everyday Objects in Everyday Scenes","date":"2016-04-12","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":30,"samples_harvested":234,"samples_ran":114,"samples_unverified":120,"pointer_only_for_licence":66,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}