{"url":"/task/video-semantic-segmentation","name":"Video Semantic Segmentation","slug":"video-semantic-segmentation","description_markdown":"The goal of video semantic segmentation is to assign a predefined class to each pixel in all frames of a video. This requires the model not only to predict accurate segmentation masks but also to ensure that these masks remain temporally consistent across frames. This task has broad applications in areas such as autonomous driving, medical video analysis, and AR/VR.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":895,"papers_with_code":418,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":1,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-semantic-segmentation-on-cityscapes-val","slug":"video-semantic-segmentation-on-cityscapes-val","dataset":"Cityscapes val","dataset_url":"/dataset/cityscapes","rows_in_archive":9,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"TMANet-50","paper_title":"Temporal Memory Attention for Video Semantic Segmentation","paper_url":"/paper/temporal-memory-attention-for-video-semantic","paper_date":"2021-02-17","arxiv_id":"2102.08643","code_links":[{"title":"wanghao9610/TMANet","url":"https://github.com/wanghao9610/TMANet"}],"syntology":null}},{"leaderboard":"/sota/video-semantic-segmentation-on-camvid","slug":"video-semantic-segmentation-on-camvid","dataset":"CamVid","dataset_url":"/dataset/camvid","rows_in_archive":6,"metrics":["Mean IoU"],"first_row_in_archive_order":{"model":"TMANet-50","paper_title":"Temporal Memory Attention for Video Semantic Segmentation","paper_url":"/paper/temporal-memory-attention-for-video-semantic","paper_date":"2021-02-17","arxiv_id":"2102.08643","code_links":[{"title":"wanghao9610/TMANet","url":"https://github.com/wanghao9610/TMANet"}],"syntology":null}},{"leaderboard":"/sota/video-semantic-segmentation-on-vspw","slug":"video-semantic-segmentation-on-vspw","dataset":"VSPW","dataset_url":"/dataset/vspw","rows_in_archive":5,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"DVIS++(VIT-L)","paper_title":"DVIS++: Improved Decoupled Framework for Universal Video Segmentation","paper_url":"/paper/dvis-improved-decoupled-framework-for","paper_date":"2023-12-20","arxiv_id":"2312.13305","code_links":[{"title":"zhang-tao-whu/DVIS_Plus","url":"https://github.com/zhang-tao-whu/DVIS_Plus"}],"syntology":null}},{"leaderboard":"/sota/video-semantic-segmentation-on-lars","slug":"video-semantic-segmentation-on-lars","dataset":"LaRS","dataset_url":"/dataset/lars","rows_in_archive":3,"metrics":["Q","F1","μ","mIoU"],"first_row_in_archive_order":{"model":"WaSR-T (ResNet-101)","paper_title":"LaRS: A Diverse Panoptic Maritime Obstacle Detection Dataset and Benchmark","paper_url":"/paper/lars-a-diverse-panoptic-maritime-obstacle","paper_date":"2023-08-18","arxiv_id":"2308.09618","code_links":[{"title":"lojzezust/lars_evaluator","url":"https://github.com/lojzezust/lars_evaluator"},{"title":"lojzezust/mmsegmentation-macvi","url":"https://github.com/lojzezust/mmsegmentation-macvi"}],"syntology":null}},{"leaderboard":"/sota/video-semantic-segmentation-on-multispectral","slug":"video-semantic-segmentation-on-multispectral","dataset":"Multispectral Video Semantic Segmentation","dataset_url":null,"rows_in_archive":3,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"MVNet(DeepLabV3)","paper_title":"Multispectral Video Semantic Segmentation: A Benchmark Dataset and Baseline","paper_url":"/paper/multispectral-video-semantic-segmentation-a","paper_date":"2023-01-01","arxiv_id":null,"code_links":[{"title":"jiwei0921/MVSS-Baseline","url":"https://github.com/jiwei0921/MVSS-Baseline"}],"syntology":null}}],"datasets":[{"url":"/dataset/cityscapes","name":"Cityscapes","full_name":"","num_papers_in_archive":3702},{"url":"/dataset/camvid","name":"CamVid","full_name":"Cambridge-driving Labeled Video Database","num_papers_in_archive":227},{"url":"/dataset/segtrack-v2-1","name":"SegTrack-v2","full_name":"","num_papers_in_archive":107},{"url":"/dataset/mose","name":"MOSE","full_name":"Complex Video Object Segmentation","num_papers_in_archive":49},{"url":"/dataset/vspw","name":"VSPW","full_name":"Video Scene Parsing in the Wild","num_papers_in_archive":35},{"url":"/dataset/lars","name":"LaRS","full_name":"Lakes, Rivers and Seas Dataset","num_papers_in_archive":5},{"url":"/dataset/vlog-dataset","name":"VLOG Dataset","full_name":"","num_papers_in_archive":4},{"url":"/dataset/odms","name":"ODMS","full_name":"Object Depth via Motion and Segmentation","num_papers_in_archive":2}],"subtasks":[{"url":"/task/camera-shot-segmentation","name":"Camera shot segmentation"}],"parent_tasks":[{"url":"/task/scene-understanding","name":"Scene Understanding"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":418,"tagged_in_all":895,"items":[{"url":"/paper/pyramid-scene-parsing-network","title":"Pyramid Scene Parsing Network","date":"2016-12-04","arxiv_id":"1612.01105","repositories_listed":67,"syntology":{"n":29,"n_ran":7,"n_unverified":22,"n_pointer_only":5}},{"url":"/paper/fully-convolutional-networks-for-semantic","title":"Fully Convolutional Networks for Semantic Segmentation","date":"2016-05-20","arxiv_id":"1605.06211","repositories_listed":37,"syntology":null},{"url":"/paper/2408-00714","title":"SAM 2: Segment Anything in Images and Videos","date":"2024-08-01","arxiv_id":"2408.00714","repositories_listed":11,"syntology":{"n":49,"n_ran":37,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/mask2former-for-video-instance-segmentation","title":"Mask2Former for Video Instance Segmentation","date":"2021-12-20","arxiv_id":"2112.10764","repositories_listed":6,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/rethinking-self-supervised-correspondence","title":"Rethinking Self-supervised Correspondence Learning: A Video Frame-level Similarity Perspective","date":"2021-03-31","arxiv_id":"2103.17263","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/modular-interactive-video-object-segmentation","title":"Modular Interactive Video Object Segmentation: Interaction-to-Mask, Propagation and Difference-Aware Fusion","date":"2021-03-14","arxiv_id":"2103.07941","repositories_listed":5,"syntology":{"n":18,"n_ran":10,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/premvos-proposal-generation-refinement-and","title":"PReMVOS: Proposal-generation, Refinement and Merging for Video Object Segmentation","date":"2018-07-24","arxiv_id":"1807.09190","repositories_listed":5,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":3}},{"url":"/paper/graphecho-graph-driven-unsupervised-domain","title":"GraphEcho: Graph-Driven Unsupervised Domain Adaptation for Echocardiogram Video Segmentation","date":"2023-09-20","arxiv_id":"2309.11145","repositories_listed":4,"syntology":null},{"url":"/paper/make-one-shot-video-object-segmentation-1","title":"Make One-Shot Video Object Segmentation Efficient Again","date":"2020-12-03","arxiv_id":"2012.01866","repositories_listed":4,"syntology":null},{"url":"/paper/interactive-video-object-segmentation-using","title":"Interactive Video Object Segmentation Using Global and Local Transfer Modules","date":"2020-07-16","arxiv_id":"2007.08139","repositories_listed":4,"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/ccnet-criss-cross-attention-for-semantic","title":"CCNet: Criss-Cross Attention for Semantic Segmentation","date":"2018-11-28","arxiv_id":"1811.11721","repositories_listed":4,"syntology":{"n":14,"n_ran":10,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/youtube-vos-sequence-to-sequence-video-object","title":"YouTube-VOS: Sequence-to-Sequence Video Object Segmentation","date":"2018-09-03","arxiv_id":"1809.00461","repositories_listed":4,"syntology":null},{"url":"/paper/lucid-data-dreaming-for-video-object","title":"Lucid Data Dreaming for Video Object Segmentation","date":"2017-03-28","arxiv_id":"1703.09554","repositories_listed":4,"syntology":null},{"url":"/paper/dvis-daq-improving-video-segmentation-via","title":"DVIS-DAQ: Improving Video Segmentation via Dynamic Anchor Queries","date":"2024-03-29","arxiv_id":"2404.00086","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":3}},{"url":"/paper/segment-everything-everywhere-all-at-once","title":"Segment Everything Everywhere All at Once","date":"2023-04-13","arxiv_id":"2304.06718","repositories_listed":3,"syntology":null},{"url":"/paper/seggpt-segmenting-everything-in-context","title":"SegGPT: Segmenting Everything In Context","date":"2023-04-06","arxiv_id":"2304.03284","repositories_listed":3,"syntology":{"n":10,"n_ran":10,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/epic-kitchens-visor-benchmark-video","title":"EPIC-KITCHENS VISOR Benchmark: VIdeo Segmentations and Object Relations","date":"2022-09-26","arxiv_id":"2209.13064","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":7}},{"url":"/paper/rethinking-space-time-networks-with-improved","title":"Rethinking Space-Time Networks with Improved Memory Coverage for Efficient Video Object Segmentation","date":"2021-06-09","arxiv_id":"2106.05210","repositories_listed":3,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":2}},{"url":"/paper/physarum-powered-differentiable-linear","title":"Physarum Powered Differentiable Linear Programming Layers and Applications","date":"2020-04-30","arxiv_id":"2004.14539","repositories_listed":3,"syntology":null},{"url":"/paper/video-object-segmentation-using-space-time","title":"Video Object Segmentation using Space-Time Memory Networks","date":"2019-04-01","arxiv_id":"1904.00607","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/feelvos-fast-end-to-end-embedding-learning","title":"FEELVOS: Fast End-to-End Embedding Learning for Video Object Segmentation","date":"2019-02-25","arxiv_id":"1902.09513","repositories_listed":3,"syntology":{"n":9,"n_ran":9,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/the-uavid-dataset-for-video-semantic","title":"UAVid: A Semantic Segmentation Dataset for UAV Imagery","date":"2018-10-24","arxiv_id":"1810.10438","repositories_listed":3,"syntology":null},{"url":"/paper/video-object-segmentation-with-re","title":"Video Object Segmentation with Re-identification","date":"2017-08-01","arxiv_id":"1708.00197","repositories_listed":3,"syntology":null},{"url":"/paper/deep-feature-flow-for-video-recognition","title":"Deep Feature Flow for Video Recognition","date":"2016-11-23","arxiv_id":"1611.07715","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/accelerating-volumetric-medical-image","title":"Accelerating Volumetric Medical Image Annotation via Short-Long Memory SAM 2","date":"2025-05-03","arxiv_id":"2505.01854","repositories_listed":2,"syntology":null},{"url":"/paper/stable-mean-teacher-for-semi-supervised-video","title":"Stable Mean Teacher for Semi-supervised Video Action Detection","date":"2024-12-10","arxiv_id":"2412.07072","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/towards-underwater-camouflaged-object","title":"Underwater Camouflaged Object Tracking Meets Vision-Language SAM2","date":"2024-09-25","arxiv_id":"2409.16902","repositories_listed":2,"syntology":null},{"url":"/paper/visa-reasoning-video-object-segmentation-via","title":"VISA: Reasoning Video Object Segmentation via Large Language Models","date":"2024-07-16","arxiv_id":"2407.11325","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/pvuw-2024-challenge-on-complex-video","title":"PVUW 2024 Challenge on Complex Video Understanding: Methods and Results","date":"2024-06-24","arxiv_id":"2406.17005","repositories_listed":2,"syntology":null},{"url":"/paper/uniref-segment-every-reference-object-in","title":"UniRef++: Segment Every Reference Object in Spatial and Temporal Spaces","date":"2023-12-25","arxiv_id":"2312.15715","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}