{"url":"/task/video-panoptic-segmentation","name":"Video Panoptic Segmentation","slug":"video-panoptic-segmentation","description_markdown":"**Video Panoptic Segmentation** is a computer vision task that extends [panoptic segmentation](https://paperswithcode.com/task/panoptic-segmentation) by incorporating temporal dimension. That is, given a video sequence, the goal is to predict the semantic class of each pixel while consistently tracking object instances. Here, the pixels belonging to the same object instance should be assigned the same instance ID throughout the video sequence.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":42,"papers_with_code":21,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/video-panoptic-segmentation-on-vipseg","slug":"video-panoptic-segmentation-on-vipseg","dataset":"VIPSeg","dataset_url":"/dataset/vipseg","rows_in_archive":12,"metrics":["VPQ","STQ"],"first_row_in_archive_order":{"model":"CAVIS(VIT-L)","paper_title":"Context-Aware Video Instance Segmentation","paper_url":"/paper/context-aware-video-instance-segmentation","paper_date":"2024-07-03","arxiv_id":"2407.03010","code_links":[{"title":"Seung-Hun-Lee/CAVIS","url":"https://github.com/Seung-Hun-Lee/CAVIS"}],"syntology":null}},{"leaderboard":"/sota/video-panoptic-segmentation-on-cityscapes-vps","slug":"video-panoptic-segmentation-on-cityscapes-vps","dataset":"Cityscapes-VPS","dataset_url":"/dataset/cityscapes-vps","rows_in_archive":8,"metrics":["VPQ","VPQ (thing)","VPQ (stuff)"],"first_row_in_archive_order":{"model":"VIP-Deeplab","paper_title":"ViP-DeepLab: Learning Visual Perception with Depth-aware Video Panoptic Segmentation","paper_url":"/paper/vip-deeplab-learning-visual-perception-with","paper_date":"2020-12-09","arxiv_id":"2012.05258","code_links":[{"title":"joe-siyuan-qiao/ViP-DeepLab","url":"https://github.com/joe-siyuan-qiao/ViP-DeepLab"},{"title":"MindSpore-scientific/code-7","url":"https://github.com/MindSpore-scientific/code-7/tree/main/GroupViT"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/video-panoptic-segmentation-on-kitti-step","slug":"video-panoptic-segmentation-on-kitti-step","dataset":"KITTI-STEP","dataset_url":"/dataset/kitti-step","rows_in_archive":6,"metrics":["AQ","SQ","STQ"],"first_row_in_archive_order":{"model":"Video K-Net (Swin-L)","paper_title":"Video K-Net: A Simple, Strong, and Unified Baseline for Video Segmentation","paper_url":"/paper/video-k-net-a-simple-strong-and-unified","paper_date":"2022-04-10","arxiv_id":"2204.04656","code_links":[{"title":"lxtgh/video-k-net","url":"https://github.com/lxtgh/video-k-net"}],"syntology":null}},{"leaderboard":"/sota/video-panoptic-segmentation-on-4d-or","slug":"video-panoptic-segmentation-on-4d-or","dataset":"4D-OR","dataset_url":"/dataset/4d-or","rows_in_archive":2,"metrics":["VPQ"],"first_row_in_archive_order":{"model":"MM-OR-VPQ4","paper_title":"MM-OR: A Large Multimodal Operating Room Dataset for Semantic Understanding of High-Intensity Surgical Environments","paper_url":"/paper/mm-or-a-large-multimodal-operating-room","paper_date":"2025-03-04","arxiv_id":"2503.02579","code_links":[{"title":"egeozsoy/MM-OR","url":"https://github.com/egeozsoy/MM-OR"}],"syntology":null}},{"leaderboard":"/sota/video-panoptic-segmentation-on-mm-or","slug":"video-panoptic-segmentation-on-mm-or","dataset":"MM-OR","dataset_url":"/dataset/mm-or","rows_in_archive":2,"metrics":["VPQ"],"first_row_in_archive_order":{"model":"MM-OR-VPQ4","paper_title":"MM-OR: A Large Multimodal Operating Room Dataset for Semantic Understanding of High-Intensity Surgical Environments","paper_url":"/paper/mm-or-a-large-multimodal-operating-room","paper_date":"2025-03-04","arxiv_id":"2503.02579","code_links":[{"title":"egeozsoy/MM-OR","url":"https://github.com/egeozsoy/MM-OR"}],"syntology":null}}],"datasets":[{"url":"/dataset/vipseg","name":"VIPSeg","full_name":"","num_papers_in_archive":32},{"url":"/dataset/cityscapes-vps","name":"Cityscapes-VPS","full_name":"","num_papers_in_archive":26},{"url":"/dataset/kitti-step","name":"KITTI-STEP","full_name":"","num_papers_in_archive":24},{"url":"/dataset/4d-or","name":"4D-OR","full_name":"","num_papers_in_archive":11},{"url":"/dataset/lars","name":"LaRS","full_name":"Lakes, Rivers and Seas Dataset","num_papers_in_archive":5},{"url":"/dataset/mm-or","name":"MM-OR","full_name":"","num_papers_in_archive":3}],"subtasks":[],"parent_tasks":[{"url":"/task/panoptic-segmentation","name":"Panoptic Segmentation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":21,"of":21,"tagged_in_all":42,"items":[{"url":"/paper/maxtron-mask-transformer-with-trajectory","title":"A Simple Video Segmenter by Tracking Objects Along Axial Trajectories","date":"2023-11-30","arxiv_id":"2311.18537","repositories_listed":2,"syntology":null},{"url":"/paper/vip-deeplab-learning-visual-perception-with","title":"ViP-DeepLab: Learning Visual Perception with Depth-aware Video Panoptic Segmentation","date":"2020-12-09","arxiv_id":"2012.05258","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/mm-or-a-large-multimodal-operating-room","title":"MM-OR: A Large Multimodal Operating Room Dataset for Semantic Understanding of High-Intensity Surgical Environments","date":"2025-03-04","arxiv_id":"2503.02579","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-video-instance-segmentation","title":"Context-Aware Video Instance Segmentation","date":"2024-07-03","arxiv_id":"2407.03010","repositories_listed":1,"syntology":null},{"url":"/paper/uni-dvps-unified-model-for-depth-aware-video","title":"Uni-DVPS: Unified Model for Depth-Aware Video Panoptic Segmentation","date":"2024-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-integrated-framework-for-multi-granular","title":"An Integrated Framework for Multi-Granular Explanation of Video Summarization","date":"2024-05-16","arxiv_id":"2405.10082","repositories_listed":1,"syntology":null},{"url":"/paper/univs-unified-and-universal-video","title":"UniVS: Unified and Universal Video Segmentation with Prompts as Queries","date":"2024-02-28","arxiv_id":"2402.18115","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_unverified":2,"n_pointer_only":14}},{"url":"/paper/dvis-improved-decoupled-framework-for","title":"DVIS++: Improved Decoupled Framework for Universal Video Segmentation","date":"2023-12-20","arxiv_id":"2312.13305","repositories_listed":1,"syntology":null},{"url":"/paper/tracking-anything-with-decoupled-video","title":"Tracking Anything with Decoupled Video Segmentation","date":"2023-09-07","arxiv_id":"2309.03903","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":10}},{"url":"/paper/1st-place-solution-for-pvuw-challenge-2023","title":"1st Place Solution for PVUW Challenge 2023: Video Panoptic Segmentation","date":"2023-06-07","arxiv_id":"2306.04091","repositories_listed":1,"syntology":null},{"url":"/paper/dvis-decoupled-video-instance-segmentation","title":"DVIS: Decoupled Video Instance Segmentation Framework","date":"2023-06-06","arxiv_id":"2306.03413","repositories_listed":1,"syntology":null},{"url":"/paper/tube-link-a-flexible-cross-tube-baseline-for","title":"Tube-Link: A Flexible Cross Tube Framework for Universal Video Segmentation","date":"2023-03-22","arxiv_id":"2303.12782","repositories_listed":1,"syntology":null},{"url":"/paper/tarvis-a-unified-approach-for-target-based","title":"TarViS: A Unified Approach for Target-based Video Segmentation","date":"2023-01-06","arxiv_id":"2301.02657","repositories_listed":1,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/context-aware-relative-object-queries-to","title":"Context-Aware Relative Object Queries To Unify Video Instance and Panoptic Segmentation","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pvo-panoptic-visual-odometry","title":"PVO: Panoptic Visual Odometry","date":"2022-07-04","arxiv_id":"2207.01610","repositories_listed":1,"syntology":null},{"url":"/paper/waymo-open-dataset-panoramic-video-panoptic","title":"Waymo Open Dataset: Panoramic Video Panoptic Segmentation","date":"2022-06-15","arxiv_id":"2206.07704","repositories_listed":1,"syntology":null},{"url":"/paper/video-k-net-a-simple-strong-and-unified","title":"Video K-Net: A Simple, Strong, and Unified Baseline for Video Segmentation","date":"2022-04-10","arxiv_id":"2204.04656","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-video-panoptic-segmentation-in","title":"Large-Scale Video Panoptic Segmentation in the Wild: A Benchmark","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/polyphonicformer-unified-query-learning-for","title":"PolyphonicFormer: Unified Query Learning for Depth-aware Video Panoptic Segmentation","date":"2021-12-05","arxiv_id":"2112.02582","repositories_listed":1,"syntology":null},{"url":"/paper/step-segmenting-and-tracking-every-pixel","title":"STEP: Segmenting and Tracking Every Pixel","date":"2021-02-23","arxiv_id":"2102.11859","repositories_listed":1,"syntology":null},{"url":"/paper/video-panoptic-segmentation-1","title":"Video Panoptic Segmentation","date":"2020-06-19","arxiv_id":"2006.11339","repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":1,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}