{"url":"/task/scene-segmentation","name":"Scene Segmentation","slug":"scene-segmentation","description_markdown":"Scene segmentation is the task of splitting a scene into its various object components.\r\n\r\nImage adapted from [Temporally coherent 4D reconstruction of complex dynamic scenes](https://paperswithcode.com/paper/temporally-coherent-4d-reconstruction-of2).","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":283,"papers_with_code":141,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":10,"subtasks":1,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/scene-segmentation-on-sun-rgbd","slug":"scene-segmentation-on-sun-rgbd","dataset":"SUN-RGBD","dataset_url":"/dataset/sun-rgb-d","rows_in_archive":5,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"ICM","paper_title":"Scene Parsing via Integrated Classification Model and Variance-Based Regularization","paper_url":"/paper/scene-parsing-via-integrated-classification","paper_date":"2019-06-01","arxiv_id":null,"code_links":[{"title":"shihengcan/ICM-matcaffe","url":"https://github.com/shihengcan/ICM-matcaffe"}],"syntology":null}},{"leaderboard":"/sota/scene-segmentation-on-scannet","slug":"scene-segmentation-on-scannet","dataset":"ScanNet","dataset_url":"/dataset/scannet","rows_in_archive":3,"metrics":["Average Accuracy","3DIoU"],"first_row_in_archive_order":{"model":"3DMV","paper_title":"3DMV: Joint 3D-Multi-View Prediction for 3D Semantic Scene Segmentation","paper_url":"/paper/3dmv-joint-3d-multi-view-prediction-for-3d","paper_date":"2018-03-28","arxiv_id":"1803.10409","code_links":[{"title":"angeladai/3DMV","url":"https://github.com/angeladai/3DMV"}],"syntology":null}},{"leaderboard":"/sota/scene-segmentation-on-streethazards","slug":"scene-segmentation-on-streethazards","dataset":"StreetHazards","dataset_url":"/dataset/streethazards","rows_in_archive":3,"metrics":["Open-mIoU"],"first_row_in_archive_order":{"model":"Mask2Anomaly","paper_title":"Unmasking Anomalies in Road-Scene Segmentation","paper_url":"/paper/unmasking-anomalies-in-road-scene","paper_date":"2023-07-25","arxiv_id":"2307.13316","code_links":[{"title":"shyam671/mask2anomaly-unmasking-anomalies-in-road-scene-segmentation","url":"https://github.com/shyam671/mask2anomaly-unmasking-anomalies-in-road-scene-segmentation"}],"syntology":null}},{"leaderboard":"/sota/scene-segmentation-on-movienet","slug":"scene-segmentation-on-movienet","dataset":"MovieNet","dataset_url":"/dataset/movienet","rows_in_archive":2,"metrics":["AP"],"first_row_in_archive_order":{"model":"NeighborNet","paper_title":"Neighbor Relations Matter in Video Scene Detection","paper_url":"/paper/neighbor-relations-matter-in-video-scene","paper_date":"2024-01-01","arxiv_id":null,"code_links":[{"title":"exmorgan-alter/neighbornet","url":"https://github.com/exmorgan-alter/neighbornet"}],"syntology":null}},{"leaderboard":"/sota/scene-segmentation-on-nyu-depth-v2","slug":"scene-segmentation-on-nyu-depth-v2","dataset":"NYU Depth v2","dataset_url":"/dataset/nyuv2","rows_in_archive":1,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"Dilated FCN-2s RGB","paper_title":"Efficient Yet Deep Convolutional Neural Networks for Semantic Segmentation","paper_url":"/paper/efficient-yet-deep-convolutional-neural","paper_date":"2017-07-26","arxiv_id":"1707.08254","code_links":[{"title":"SharifAmit/DilatedFCNSegmentation","url":"https://github.com/SharifAmit/DilatedFCNSegmentation"}],"syntology":null}},{"leaderboard":"/sota/scene-segmentation-on-uavid","slug":"scene-segmentation-on-uavid","dataset":"UAVid","dataset_url":"/dataset/uavid","rows_in_archive":1,"metrics":["Category mIoU"],"first_row_in_archive_order":{"model":"UNetFormer","paper_title":"UNetFormer: A UNet-like Transformer for Efficient Semantic Segmentation of Remote Sensing Urban Scene Imagery","paper_url":"/paper/efficient-hybrid-transformer-learning-global","paper_date":"2021-09-18","arxiv_id":"2109.08937","code_links":[{"title":"WangLibo1995/GeoSeg","url":"https://github.com/WangLibo1995/GeoSeg"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}}],"datasets":[{"url":"/dataset/scannet","name":"ScanNet","full_name":"","num_papers_in_archive":1595},{"url":"/dataset/nyuv2","name":"NYUv2","full_name":"NYU-Depth V2","num_papers_in_archive":986},{"url":"/dataset/sun-rgb-d","name":"SUN RGB-D","full_name":"SUN RGB-D","num_papers_in_archive":477},{"url":"/dataset/movienet","name":"MovieNet","full_name":"MovieNet","num_papers_in_archive":54},{"url":"/dataset/uavid","name":"UAVid","full_name":"","num_papers_in_archive":54},{"url":"/dataset/streethazards","name":"StreetHazards","full_name":"","num_papers_in_archive":20},{"url":"/dataset/usis10k","name":"USIS10K","full_name":"Large-scale Underwater Salient Instance Segmentation Dataset","num_papers_in_archive":4},{"url":"/dataset/mila-simulated-floods","name":"Mila Simulated Floods","full_name":"","num_papers_in_archive":2},{"url":"/dataset/berkeley-deepdrive-video","name":"Berkeley DeepDrive Video","full_name":"","num_papers_in_archive":1},{"url":"/dataset/darai","name":"DARai","full_name":"Daily Activity Recordings for AI and ML applications","num_papers_in_archive":1}],"subtasks":[{"url":"/task/thermal-image-segmentation","name":"Thermal Image Segmentation"}],"parent_tasks":[{"url":"/task/semantic-segmentation","name":"Semantic Segmentation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":141,"tagged_in_all":283,"items":[{"url":"/paper/pointnet-deep-learning-on-point-sets-for-3d","title":"PointNet: Deep Learning on Point Sets for 3D Classification and Segmentation","date":"2016-12-02","arxiv_id":"1612.00593","repositories_listed":110,"syntology":{"n":164,"n_ran":89,"n_unverified":75,"n_pointer_only":90}},{"url":"/paper/segnet-a-deep-convolutional-encoder-decoder","title":"SegNet: A Deep Convolutional Encoder-Decoder Architecture for Image Segmentation","date":"2015-11-02","arxiv_id":"1511.00561","repositories_listed":74,"syntology":{"n":44,"n_ran":9,"n_unverified":35,"n_pointer_only":10}},{"url":"/paper/fully-convolutional-networks-for-semantic","title":"Fully Convolutional Networks for Semantic Segmentation","date":"2016-05-20","arxiv_id":"1605.06211","repositories_listed":37,"syntology":null},{"url":"/paper/point-transformer-1","title":"Point Transformer","date":"2020-12-16","arxiv_id":"2012.09164","repositories_listed":24,"syntology":null},{"url":"/paper/dual-attention-network-for-scene-segmentation","title":"Dual Attention Network for Scene Segmentation","date":"2018-09-09","arxiv_id":"1809.02983","repositories_listed":12,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":5}},{"url":"/paper/kpconv-flexible-and-deformable-convolution","title":"KPConv: Flexible and Deformable Convolution for Point Clouds","date":"2019-04-18","arxiv_id":"1904.08889","repositories_listed":10,"syntology":{"n":12,"n_ran":5,"n_unverified":7,"n_pointer_only":3}},{"url":"/paper/panoptic-segmentation","title":"Panoptic Segmentation","date":"2018-01-03","arxiv_id":"1801.00868","repositories_listed":9,"syntology":null},{"url":"/paper/index-network","title":"Index Network","date":"2019-08-11","arxiv_id":"1908.09895","repositories_listed":6,"syntology":null},{"url":"/paper/seamless-scene-segmentation","title":"Seamless Scene Segmentation","date":"2019-05-03","arxiv_id":"1905.01220","repositories_listed":5,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/a-local-to-global-approach-to-multi-modal","title":"A Local-to-Global Approach to Multi-modal Movie Scene Segmentation","date":"2020-04-06","arxiv_id":"2004.02678","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/point-voxel-cnn-for-efficient-3d-deep","title":"Point-Voxel CNN for Efficient 3D Deep Learning","date":"2019-07-08","arxiv_id":"1907.03739","repositories_listed":4,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":5}},{"url":"/paper/no-time-to-train-empowering-non-parametric","title":"No Time to Train: Empowering Non-Parametric Networks for Few-shot 3D Scene Segmentation","date":"2024-04-05","arxiv_id":"2404.04050","repositories_listed":3,"syntology":{"n":14,"n_ran":8,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/eprnet-efficient-pyramid-representation","title":"EPRNet: Efficient Pyramid Representation Network for Real-Time Street Scene Segmentation","date":"2021-03-23","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/2018-robotic-scene-segmentation-challenge","title":"2018 Robotic Scene Segmentation Challenge","date":"2020-01-30","arxiv_id":"2001.11190","repositories_listed":3,"syntology":null},{"url":"/paper/a-benchmark-for-endoluminal-scene","title":"A Benchmark for Endoluminal Scene Segmentation of Colonoscopy Images","date":"2016-12-02","arxiv_id":"1612.00799","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/transferring-to-real-world-layouts-a-depth","title":"Transferring to Real-World Layouts: A Depth-aware Framework for Scene Adaptation","date":"2023-11-21","arxiv_id":"2311.12682","repositories_listed":2,"syntology":null},{"url":"/paper/double-domain-guided-real-time-low-light","title":"Double Domain Guided Real-Time Low-Light Image Enhancement for Ultra-High-Definition Transportation Surveillance","date":"2023-09-15","arxiv_id":"2309.08382","repositories_listed":2,"syntology":null},{"url":"/paper/neural-implicit-vision-language-feature","title":"Neural Implicit Vision-Language Feature Fields","date":"2023-03-20","arxiv_id":"2303.10962","repositories_listed":2,"syntology":null},{"url":"/paper/residual-pattern-learning-for-pixel-wise-out","title":"Residual Pattern Learning for Pixel-wise Out-of-Distribution Detection in Semantic Segmentation","date":"2022-11-26","arxiv_id":"2211.14512","repositories_listed":2,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/improving-nighttime-driving-scene","title":"Improving Nighttime Driving-Scene Segmentation via Dual Image-adaptive Learnable Filters","date":"2022-07-04","arxiv_id":"2207.01331","repositories_listed":2,"syntology":null},{"url":"/paper/rethinking-surgical-instrument-segmentation-a","title":"Rethinking Surgical Instrument Segmentation: A Background Image Can Be All You Need","date":"2022-06-23","arxiv_id":"2206.11804","repositories_listed":2,"syntology":null},{"url":"/paper/bevfusion-multi-task-multi-sensor-fusion-with","title":"BEVFusion: Multi-Task Multi-Sensor Fusion with Unified Bird's-Eye View Representation","date":"2022-05-26","arxiv_id":"2205.13542","repositories_listed":2,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/fifo-learning-fog-invariant-features-for-1","title":"FIFO: Learning Fog-invariant Features for Foggy Scene Segmentation","date":"2022-04-04","arxiv_id":"2204.01587","repositories_listed":2,"syntology":null},{"url":"/paper/geometric-feature-learning-for-3d-meshes","title":"Mesh Convolution with Continuous Filters for 3D Surface Parsing","date":"2021-12-03","arxiv_id":"2112.01801","repositories_listed":2,"syntology":null},{"url":"/paper/condnet-conditional-classifier-for-scene","title":"CondNet: Conditional Classifier for Scene Segmentation","date":"2021-09-21","arxiv_id":"2109.10322","repositories_listed":2,"syntology":null},{"url":"/paper/adaptive-boosting-for-domain-adaptation","title":"Adaptive Boosting for Domain Adaptation: Towards Robust Predictions in Scene Segmentation","date":"2021-03-29","arxiv_id":"2103.15685","repositories_listed":2,"syntology":null},{"url":"/paper/robustnet-improving-domain-generalization-in","title":"RobustNet: Improving Domain Generalization in Urban-Scene Segmentation via Instance Selective Whitening","date":"2021-03-29","arxiv_id":"2103.15597","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/learning-and-reasoning-with-the-graph","title":"Learning and Reasoning with the Graph Structure Representation in Robotic Surgery","date":"2020-07-07","arxiv_id":"2007.03357","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":8}},{"url":"/paper/context-prior-for-scene-segmentation","title":"Context Prior for Scene Segmentation","date":"2020-04-03","arxiv_id":"2004.01547","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-learning-of-driving-models-from","title":"End-to-end Learning of Driving Models from Large-scale Video Datasets","date":"2016-12-04","arxiv_id":"1612.01079","repositories_listed":2,"syntology":null}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}