{"url":"/dataset/cityscapes-vps","name":"Cityscapes-VPS","full_name":null,"description_markdown":"Cityscapes-VPS is a video extension of the Cityscapes validation split. It provides 2500-frame panoptic labels that temporally extend the 500 Cityscapes image-panoptic labels. There are total 3000-frame panoptic labels which correspond to 5, 10, 15, 20, 25, and 30th frames of each 500 videos, where all instance ids are associated over time. It not only supports video panoptic segmentation (VPS) task, but also provides super-set annotations for video semantic segmentation (VSS) and video instance segmentation (VIS) tasks.","description_withheld":null,"homepage":"https://github.com/mcahny/vps","introduced_date":"2020-06-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/video-panoptic-segmentation-1","title":"Video Panoptic Segmentation","first_author":"Dahun Kim","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Video Panoptic Segmentation","url":"/task/video-panoptic-segmentation","datasets_with_task":"/datasets/task/video-panoptic-segmentation"}],"languages":[],"variants":["Cityscapes-VPS"],"data_loaders":[{"repo":"https://github.com/mcahny/vps","url":"https://github.com/mcahny/vps","frameworks":["pytorch"]}],"num_papers_in_archive":26,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-panoptic-segmentation-on-cityscapes-vps","task":"Video Panoptic Segmentation","dataset_variant":"Cityscapes-VPS","rows":8,"metrics":["VPQ","VPQ (thing)","VPQ (stuff)"],"first_row_in_archive_order":{"model":"VIP-Deeplab","paper":"/paper/vip-deeplab-learning-visual-perception-with","metrics":{"VPQ":"63.1","VPQ (stuff)":"73.0","VPQ (thing)":"49.5"},"code_links":[{"title":"joe-siyuan-qiao/ViP-DeepLab","url":"https://github.com/joe-siyuan-qiao/ViP-DeepLab"},{"title":"MindSpore-scientific/code-7","url":"https://github.com/MindSpore-scientific/code-7/tree/main/GroupViT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/tarvis-a-unified-approach-for-target-based","title":"TarViS: A Unified Approach for Target-based Video Segmentation","date":"2023-01-06","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-k-net-a-simple-strong-and-unified","title":"Video K-Net: A Simple, Strong, and Unified Baseline for Video Segmentation","date":"2022-04-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/polyphonicformer-unified-query-learning-for","title":"PolyphonicFormer: Unified Query Learning for Depth-aware Video Panoptic Segmentation","date":"2021-12-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-to-associate-every-segment-for-video","title":"Learning to Associate Every Segment for Video Panoptic Segmentation","date":"2021-06-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vip-deeplab-learning-visual-perception-with","title":"ViP-DeepLab: Learning Visual Perception with Depth-aware Video Panoptic Segmentation","date":"2020-12-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-panoptic-segmentation-1","title":"Video Panoptic Segmentation","date":"2020-06-19","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}