{"url":"/task/scene-parsing","name":"Scene Parsing","slug":"scene-parsing","description_markdown":"Scene parsing is to segment and parse an image into different image regions associated with semantic categories, such as sky, road, person, and bed. [MIT Description](http://sceneparsing.csail.mit.edu/#:~:text=Scene%20parsing%20is%20to%20segment,the%20algorithms%20of%20scene%20parsing.)","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":199,"papers_with_code":80,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":9,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/scene-parsing-on-pgdp5k","slug":"scene-parsing-on-pgdp5k","dataset":"PGDP5K","dataset_url":"/dataset/pgdp5k","rows_in_archive":2,"metrics":["Total Accuracy"],"first_row_in_archive_order":{"model":"PGDPNet","paper_title":"Plane Geometry Diagram Parsing","paper_url":"/paper/plane-geometry-diagram-parsing","paper_date":"2022-05-19","arxiv_id":"2205.09363","code_links":[{"title":"mingliangzhang2018/PGDP","url":"https://github.com/mingliangzhang2018/PGDP"}],"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/scene-parsing-on-cityscapes-test","slug":"scene-parsing-on-cityscapes-test","dataset":"Cityscapes test","dataset_url":"/dataset/cityscapes","rows_in_archive":1,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"VCD No Coarse","paper_title":"Variational Context-Deformable ConvNets for Indoor Scene Parsing","paper_url":"/paper/variational-context-deformable-convnets-for","paper_date":"2020-06-01","arxiv_id":null,"code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/cityscapes","name":"Cityscapes","full_name":"","num_papers_in_archive":3702},{"url":"/dataset/stanford-background","name":"Stanford Background","full_name":"Standford Background Dataset","num_papers_in_archive":46},{"url":"/dataset/pgdp5k","name":"PGDP5K","full_name":"Plane Geometry Diagram Parsing Dataset","num_papers_in_archive":5},{"url":"/dataset/undd","name":"UNDD","full_name":"Urban Night Driving Dataset","num_papers_in_archive":3}],"subtasks":[{"url":"/task/face-parsing","name":"Face Parsing"},{"url":"/task/indoor-scene-reconstruction","name":"Indoor Scene Reconstruction"},{"url":"/task/indoor-scene-synthesis","name":"Indoor Scene Synthesis"},{"url":"/task/scene-graph-generation","name":"Scene Graph Generation"},{"url":"/task/scene-labeling","name":"Scene Labeling"},{"url":"/task/scene-recognition","name":"Scene Recognition"},{"url":"/task/scene-text-recognition","name":"Scene Text Recognition"},{"url":"/task/scene-understanding","name":"Scene Understanding"},{"url":"/task/street-scene-parsing","name":"Street Scene Parsing"}],"parent_tasks":[{"url":"/task/2d-semantic-segmentation","name":"2D Semantic Segmentation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":80,"tagged_in_all":199,"items":[{"url":"/paper/pyramid-scene-parsing-network","title":"Pyramid Scene Parsing Network","date":"2016-12-04","arxiv_id":"1612.01105","repositories_listed":67,"syntology":{"n":29,"n_ran":7,"n_unverified":22,"n_pointer_only":5}},{"url":"/paper/semantic-understanding-of-scenes-through-the","title":"Semantic Understanding of Scenes through the ADE20K Dataset","date":"2016-08-18","arxiv_id":"1608.05442","repositories_listed":22,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/panoptic-segmentation","title":"Panoptic Segmentation","date":"2018-01-03","arxiv_id":"1801.00868","repositories_listed":9,"syntology":null},{"url":"/paper/deep-dual-resolution-networks-for-real-time","title":"Deep Dual-resolution Networks for Real-time and Accurate Semantic Segmentation of Road Scenes","date":"2021-01-15","arxiv_id":"2101.06085","repositories_listed":8,"syntology":null},{"url":"/paper/ocnet-object-context-network-for-scene","title":"OCNet: Object Context Network for Scene Parsing","date":"2018-09-04","arxiv_id":"1809.00916","repositories_listed":8,"syntology":null},{"url":"/paper/resunet-a-a-deep-learning-framework-for","title":"ResUNet-a: a deep learning framework for semantic segmentation of remotely sensed data","date":"2019-04-01","arxiv_id":"1904.00592","repositories_listed":7,"syntology":{"n":13,"n_ran":1,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/semantic-flow-for-fast-and-accurate-scene","title":"Semantic Flow for Fast and Accurate Scene Parsing","date":"2020-02-24","arxiv_id":"2002.10120","repositories_listed":6,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":1}},{"url":"/paper/oneformer-one-transformer-to-rule-universal","title":"OneFormer: One Transformer to Rule Universal Image Segmentation","date":"2022-11-10","arxiv_id":"2211.06220","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/psanet-point-wise-spatial-attention-network","title":"PSANet: Point-wise Spatial Attention Network for Scene Parsing","date":"2018-09-01","arxiv_id":null,"repositories_listed":4,"syntology":null},{"url":"/paper/pyramidal-convolution-rethinking","title":"Pyramidal Convolution: Rethinking Convolutional Neural Networks for Visual Recognition","date":"2020-06-20","arxiv_id":"2006.11538","repositories_listed":3,"syntology":null},{"url":"/paper/a-dense-material-segmentation-dataset-for","title":"A Dense Material Segmentation Dataset for Indoor and Outdoor Scene Parsing","date":"2022-07-21","arxiv_id":"2207.10614","repositories_listed":2,"syntology":null},{"url":"/paper/geometric-feature-learning-for-3d-meshes","title":"Mesh Convolution with Continuous Filters for 3D Surface Parsing","date":"2021-12-03","arxiv_id":"2112.01801","repositories_listed":2,"syntology":null},{"url":"/paper/kimera-from-slam-to-spatial-perception-with","title":"Kimera: from SLAM to Spatial Perception with 3D Dynamic Scene Graphs","date":"2021-01-18","arxiv_id":"2101.06894","repositories_listed":2,"syntology":null},{"url":"/paper/minimal-solvers-for-single-view-lens","title":"Minimal Solvers for Single-View Lens-Distorted Camera Auto-Calibration","date":"2020-11-17","arxiv_id":"2011.08988","repositories_listed":2,"syntology":null},{"url":"/paper/malleable-2-5d-convolution-learning-receptive","title":"Malleable 2.5D Convolution: Learning Receptive Fields along the Depth-axis for RGB-D Scene Parsing","date":"2020-07-18","arxiv_id":"2007.09365","repositories_listed":2,"syntology":{"n":14,"n_ran":4,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/cascadepsp-toward-class-agnostic-and-very","title":"CascadePSP: Toward Class-Agnostic and Very High-Resolution Segmentation via Global and Local Refinement","date":"2020-05-06","arxiv_id":"2005.02551","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/2003-13328","title":"Strip Pooling: Rethinking Spatial Pooling for Scene Parsing","date":"2020-03-30","arxiv_id":"2003.13328","repositories_listed":2,"syntology":{"n":11,"n_ran":2,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/dynamic-multi-scale-filters-for-semantic","title":"Dynamic Multi-Scale Filters for Semantic Segmentation","date":"2019-10-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/slimyolov3-narrower-faster-and-better-for","title":"SlimYOLOv3: Narrower, Faster and Better for Real-Time UAV Applications","date":"2019-07-25","arxiv_id":"1907.11093","repositories_listed":2,"syntology":null},{"url":"/paper/gff-gated-fully-fusion-for-semantic","title":"GFF: Gated Fully Fusion for Semantic Segmentation","date":"2019-04-03","arxiv_id":"1904.01803","repositories_listed":2,"syntology":null},{"url":"/paper/context-aware-synthesis-and-placement-of","title":"Context-Aware Synthesis and Placement of Object Instances","date":"2018-12-06","arxiv_id":"1812.02350","repositories_listed":2,"syntology":null},{"url":"/paper/synscapes-a-photorealistic-synthetic-dataset","title":"Synscapes: A Photorealistic Synthetic Dataset for Street Scene Parsing","date":"2018-10-19","arxiv_id":"1810.08705","repositories_listed":2,"syntology":null},{"url":"/paper/multi-grained-contrast-for-data-efficient","title":"Multi-Grained Contrast for Data-Efficient Unsupervised Representation Learning","date":"2024-07-02","arxiv_id":"2407.02014","repositories_listed":1,"syntology":null},{"url":"/paper/pig-prompt-images-guidance-for-night-time","title":"PIG: Prompt Images Guidance for Night-Time Scene Parsing","date":"2024-06-15","arxiv_id":"2406.10531","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-fruit-segmentation-via-transfer","title":"Few-Shot Fruit Segmentation via Transfer Learning","date":"2024-05-04","arxiv_id":"2405.02556","repositories_listed":1,"syntology":null},{"url":"/paper/hapnet-toward-superior-rgb-thermal-scene","title":"HAPNet: Toward Superior RGB-Thermal Scene Parsing via Hybrid, Asymmetric, and Progressive Heterogeneous Feature Fusion","date":"2024-04-04","arxiv_id":"2404.03527","repositories_listed":1,"syntology":null},{"url":"/paper/robust-shape-fitting-for-3d-scene-abstraction","title":"Robust Shape Fitting for 3D Scene Abstraction","date":"2024-03-15","arxiv_id":"2403.10452","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-semantic-segmentation-of-high","title":"Applying Unsupervised Semantic Segmentation to High-Resolution UAV Imagery for Enhanced Road Scene Parsing","date":"2024-02-05","arxiv_id":"2402.02985","repositories_listed":1,"syntology":null},{"url":"/paper/lf-tracy-a-unified-single-pipeline-approach","title":"LF Tracy: A Unified Single-Pipeline Approach for Salient Object Detection in Light Field Cameras","date":"2024-01-30","arxiv_id":"2401.16712","repositories_listed":1,"syntology":null},{"url":"/paper/a-data-efficient-framework-for-robotics-large","title":"A Data-efficient Framework for Robotics Large-scale LiDAR Scene Parsing","date":"2023-12-03","arxiv_id":"2312.02208","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}