{"url":"/task/zero-shot-segmentation","name":"Zero Shot Segmentation","slug":"zero-shot-segmentation","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":134,"papers_with_code":70,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/zero-shot-segmentation-on-segmentation-in-the","slug":"zero-shot-segmentation-on-segmentation-in-the","dataset":"Segmentation in the Wild","dataset_url":"/dataset/segmentation-in-the-wild","rows_in_archive":12,"metrics":["Mean AP"],"first_row_in_archive_order":{"model":"Grounded HQ-SAM","paper_title":"Segment Anything in High Quality","paper_url":"/paper/segment-anything-in-high-quality","paper_date":"2023-06-02","arxiv_id":"2306.01567","code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"IDEA-Research/Grounded-Segment-Anything","url":"https://github.com/IDEA-Research/Grounded-Segment-Anything"},{"title":"syscv/sam-hq","url":"https://github.com/syscv/sam-hq"},{"title":"sqhuang0103/samreg","url":"https://github.com/sqhuang0103/samreg"}],"syntology":{"n":17,"n_ran":3,"n_unverified":14,"n_pointer_only":3}}},{"leaderboard":"/sota/zero-shot-segmentation-on-ade20k-training","slug":"zero-shot-segmentation-on-ade20k-training","dataset":"ADE20K training-free zero-shot segmentation","dataset_url":null,"rows_in_archive":5,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"COSMOS ViT-B/16","paper_title":"COSMOS: Cross-Modality Self-Distillation for Vision Language Pre-training","paper_url":"/paper/cosmos-cross-modality-self-distillation-for","paper_date":"2024-12-02","arxiv_id":"2412.01814","code_links":[{"title":"ExplainableML/cosmos","url":"https://github.com/ExplainableML/cosmos"}],"syntology":null}}],"datasets":[{"url":"/dataset/segmentation-in-the-wild","name":"Segmentation in the Wild","full_name":"Segmentation in the Wild","num_papers_in_archive":15},{"url":"/dataset/matseg-dataset-for-zero-shot-material-states","name":"MatSeg","full_name":"Dataset for Zero-Shot Material States Segmentation","num_papers_in_archive":1},{"url":"/dataset/tomosam","name":"TomoSAM","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":70,"tagged_in_all":134,"items":[{"url":"/paper/grounding-dino-marrying-dino-with-grounded","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","date":"2023-03-09","arxiv_id":"2303.05499","repositories_listed":10,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/prompt-based-multi-modal-image-segmentation","title":"Image Segmentation Using Text and Image Prompts","date":"2021-12-18","arxiv_id":"2112.10003","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/segment-anything-in-high-quality","title":"Segment Anything in High Quality","date":"2023-06-02","arxiv_id":"2306.01567","repositories_listed":4,"syntology":{"n":17,"n_ran":3,"n_unverified":14,"n_pointer_only":3}},{"url":"/paper/seg-zero-reasoning-chain-guided-segmentation","title":"Seg-Zero: Reasoning-Chain Guided Segmentation via Cognitive Reinforcement","date":"2025-03-09","arxiv_id":"2503.06520","repositories_listed":3,"syntology":null},{"url":"/paper/promptunet-toward-interactive-medical-image","title":"One-Prompt to Segment All Medical Images","date":"2023-05-17","arxiv_id":"2305.10300","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/side-adapter-network-for-open-vocabulary","title":"Side Adapter Network for Open-Vocabulary Semantic Segmentation","date":"2023-02-23","arxiv_id":"2302.12242","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/medclip-samv2-towards-universal-text-driven","title":"MedCLIP-SAMv2: Towards Universal Text-Driven Medical Image Segmentation","date":"2024-09-28","arxiv_id":"2409.19483","repositories_listed":2,"syntology":null},{"url":"/paper/learning-mask-aware-clip-representations-for","title":"Learning Mask-aware CLIP Representations for Zero-Shot Segmentation","date":"2023-09-30","arxiv_id":"2310.00240","repositories_listed":2,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":2}},{"url":"/paper/segment-anything-model-for-medical-image","title":"Segment Anything Model for Medical Image Analysis: an Experimental Study","date":"2023-04-20","arxiv_id":"2304.10517","repositories_listed":2,"syntology":null},{"url":"/paper/clip-surgery-for-better-explainability-with","title":"A Closer Look at the Explainability of Contrastive Language-Image Pre-training","date":"2023-04-12","arxiv_id":"2304.05653","repositories_listed":2,"syntology":null},{"url":"/paper/a-simple-framework-for-open-vocabulary","title":"A Simple Framework for Open-Vocabulary Segmentation and Detection","date":"2023-03-14","arxiv_id":"2303.08131","repositories_listed":2,"syntology":null},{"url":"/paper/context-aware-feature-generation-for-zero","title":"Context-aware Feature Generation for Zero-shot Semantic Segmentation","date":"2020-08-16","arxiv_id":"2008.06893","repositories_listed":2,"syntology":null},{"url":"/paper/compress-any-segment-anything-model-sam","title":"Compress Any Segment Anything Model (SAM)","date":"2025-07-11","arxiv_id":"2507.08765","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-tree-detection-and-segmentation","title":"Zero-Shot Tree Detection and Segmentation from Aerial Forest Imagery","date":"2025-06-03","arxiv_id":"2506.03114","repositories_listed":1,"syntology":null},{"url":"/paper/removing-watermarks-with-partial-regeneration","title":"Removing Watermarks with Partial Regeneration using Semantic Information","date":"2025-05-13","arxiv_id":"2505.08234","repositories_listed":1,"syntology":null},{"url":"/paper/3d-pointzshots-geometry-aware-3d-point-cloud","title":"3D-PointZshotS: Geometry-Aware 3D Point Cloud Zero-Shot Semantic Segmentation Narrowing the Visual-Semantic Gap","date":"2025-04-16","arxiv_id":"2504.12442","repositories_listed":1,"syntology":null},{"url":"/paper/cellvit-energy-efficient-and-adaptive-cell","title":"CellViT++: Energy-Efficient and Adaptive Cell Segmentation and Classification Using Foundation Models","date":"2025-01-09","arxiv_id":"2501.05269","repositories_listed":1,"syntology":null},{"url":"/paper/hola-hololens-object-labeling","title":"HOLa: HoloLens Object Labeling","date":"2024-12-06","arxiv_id":"2412.04945","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-the-limits-of-segment-anything","title":"Quantifying the Limits of Segmentation Foundation Models: Modeling Challenges in Segmenting Tree-Like and Low-Contrast Objects","date":"2024-12-05","arxiv_id":"2412.04243","repositories_listed":1,"syntology":null},{"url":"/paper/cosmos-cross-modality-self-distillation-for","title":"COSMOS: Cross-Modality Self-Distillation for Vision Language Pre-training","date":"2024-12-02","arxiv_id":"2412.01814","repositories_listed":1,"syntology":null},{"url":"/paper/3dgs-cd-3d-gaussian-splatting-based-change","title":"3DGS-CD: 3D Gaussian Splatting-based Change Detection for Physical Object Rearrangement","date":"2024-11-06","arxiv_id":"2411.03706","repositories_listed":1,"syntology":null},{"url":"/paper/zim-zero-shot-image-matting-for-anything","title":"ZIM: Zero-Shot Image Matting for Anything","date":"2024-11-01","arxiv_id":"2411.00626","repositories_listed":1,"syntology":null},{"url":"/paper/interpreting-and-editing-vision-language","title":"Interpreting and Editing Vision-Language Representations to Mitigate Hallucinations","date":"2024-10-03","arxiv_id":"2410.02762","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/evaluation-study-on-sam-2-for-class-agnostic","title":"Evaluation Study on SAM 2 for Class-agnostic Instance-level Segmentation","date":"2024-09-04","arxiv_id":"2409.02567","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-potential-of-sam2-for","title":"Unleashing the Potential of SAM2 for Biomedical Images and Videos: A Survey","date":"2024-08-23","arxiv_id":"2408.12889","repositories_listed":1,"syntology":null},{"url":"/paper/sam-unet-enhancing-zero-shot-segmentation-of","title":"SAM-UNet:Enhancing Zero-Shot Segmentation of SAM for Universal Medical Images","date":"2024-08-19","arxiv_id":"2408.09886","repositories_listed":1,"syntology":null},{"url":"/paper/pavecap-the-first-multimodal-framework-for","title":"PaveCap: The First Multimodal Framework for Comprehensive Pavement Condition Assessment with Dense Captioning and PCI Estimation","date":"2024-08-07","arxiv_id":"2408.04110","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01648","title":"Zero-Shot Surgical Tool Segmentation in Monocular Video Using Segment Anything Model 2","date":"2024-08-03","arxiv_id":"2408.01648","repositories_listed":1,"syntology":null},{"url":"/paper/x-recon-learning-based-patient-specific-high","title":"X-Recon: Learning-based Patient-specific High-Resolution CT Reconstruction from Orthogonal X-Ray Images","date":"2024-07-22","arxiv_id":"2407.15356","repositories_listed":1,"syntology":null},{"url":"/paper/meshsegmenter-zero-shot-mesh-semantic","title":"MeshSegmenter: Zero-Shot Mesh Semantic Segmentation via Texture Synthesis","date":"2024-07-18","arxiv_id":"2407.13675","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}