{"url":"/dataset/visual-perception-viper","name":"VIsual PERception (VIPER)","full_name":null,"description_markdown":"VIPER is a benchmark suite for visual perception. The benchmark is based on more than 250K high-resolution video frames, all annotated with ground-truth data for both low-level and high-level vision tasks, including optical flow, semantic instance segmentation, object detection and tracking, object-level 3D scene layout, and visual odometry. Ground-truth data for all tasks is available for every frame. The data was collected while driving, riding, and walking a total of 184 kilometers in diverse ambient conditions in a realistic virtual world. \r\n\r\nSource: [Playing for Benchmarks](/paper/playing-for-benchmarks)\r\n\r\nImage Source: [Playing for Benchmarks](/paper/playing-for-benchmarks)","description_withheld":null,"homepage":"https://playing-for-benchmarks.org/overview/","introduced_date":"2017-09-21","introduced_date_note":null,"introduced_by":{"paper":"/paper/playing-for-benchmarks","title":"Playing for Benchmarks","first_author":"Stephan R. Richter","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"3D","url":"/datasets/modality/3d"}],"tasks":[],"languages":[],"variants":["VIsual PERception (VIPER)"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}