{"url":"/dataset/kitti-step","name":"KITTI-STEP","full_name":null,"description_markdown":"The Segmenting and Tracking Every Pixel (STEP) benchmark consists of 21 training sequences and 29 test sequences. It is based on the KITTI Tracking Evaluation and the Multi-Object Tracking and Segmentation (MOTS) benchmark. This benchmark extends the annotations to the Segmenting and Tracking Every Pixel (STEP) task. [Copy-pasted from http://www.cvlibs.net/datasets/kitti/eval_step.php]","description_withheld":null,"homepage":"http://www.cvlibs.net/datasets/kitti//eval_step.php","introduced_date":"2021-02-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/step-segmenting-and-tracking-every-pixel","title":"STEP: Segmenting and Tracking Every Pixel","first_author":"Mark Weber","url":null},"license":{"name":"CC BY-NC-SA 3.0","url":"http://www.cvlibs.net/datasets/kitti/index.php"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Unsupervised Semantic Segmentation","url":"/task/unsupervised-semantic-segmentation","datasets_with_task":"/datasets/task/unsupervised-semantic-segmentation"},{"name":"Unsupervised Semantic Segmentation with Language-image Pre-training","url":"/task/unsupervised-semantic-segmentation-with","datasets_with_task":"/datasets/task/unsupervised-semantic-segmentation-with"},{"name":"Video Panoptic Segmentation","url":"/task/video-panoptic-segmentation","datasets_with_task":"/datasets/task/video-panoptic-segmentation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["KITTI-STEP"],"data_loaders":[],"num_papers_in_archive":24,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-panoptic-segmentation-on-kitti-step","task":"Video Panoptic Segmentation","dataset_variant":"KITTI-STEP","rows":6,"metrics":["AQ","SQ","STQ"],"first_row_in_archive_order":{"model":"Video K-Net (Swin-L)","paper":"/paper/video-k-net-a-simple-strong-and-unified","metrics":{"AQ":"73.0","SQ":"75.0","STQ":"74.0"},"code_links":[{"title":"lxtgh/video-k-net","url":"https://github.com/lxtgh/video-k-net"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-semantic-segmentation-with-2","task":"Unsupervised Semantic Segmentation with Language-image Pre-training","dataset_variant":"KITTI-STEP","rows":3,"metrics":["mIoU","pixel accuracy"],"first_row_in_archive_order":{"model":"ReCo+","paper":"/paper/reco-retrieve-and-co-segment-for-zero-shot-1","metrics":{"mIoU":"31.9","pixel accuracy":"75.3"},"code_links":[{"title":"NoelShin/reco","url":"https://github.com/NoelShin/reco"},{"title":"noelshin/namedmask","url":"https://github.com/noelshin/namedmask"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/tube-link-a-flexible-cross-tube-baseline-for","title":"Tube-Link: A Flexible Cross Tube Framework for Universal Video Segmentation","date":"2023-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unified-perception-efficient-video-panoptic","title":"Unified Perception: Efficient Depth-Aware Video Panoptic Segmentation with Minimal Annotation Costs","date":"2023-03-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tarvis-a-unified-approach-for-target-based","title":"TarViS: A Unified Approach for Target-based Video Segmentation","date":"2023-01-06","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reco-retrieve-and-co-segment-for-zero-shot-1","title":"ReCo: Retrieve and Co-segment for Zero-shot Transfer","date":"2022-06-14","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/video-k-net-a-simple-strong-and-unified","title":"Video K-Net: A Simple, Strong, and Unified Baseline for Video Segmentation","date":"2022-04-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/denseclip-extract-free-dense-labels-from-clip","title":"Extract Free Dense Labels from CLIP","date":"2021-12-02","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}