{"url":"/dataset/perception-test","name":"Perception Test","full_name":null,"description_markdown":"Perception Test is a benchmark designed to evaluate the perception and reasoning skills of multimodal models. It introduces real-world videos designed to show perceptually interesting situations and defines multiple tasks that require understanding of memory, abstract patterns, physics, and semantics – across visual, audio, and text modalities. The benchmark consists of 11.6k videos, 23s average length, filmed by around 100 participants worldwide. The videos are densely annotated with six types of labels: object and point tracks, temporal action and sound segments, multiple-choice video question-answers and grounded video question-answers. The benchmark probes pre-trained models for their transfer capabilities, in a zero-shot / few-shot or fine tuning regime.","description_withheld":null,"homepage":"https://github.com/deepmind/perception_test","introduced_date":"2022-10-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/perception-test-a-diagnostic-benchmark-for","title":"Perception Test: A Diagnostic Benchmark for Multimodal Models","first_author":"Viorica Pătrăucean","url":null},"license":{"name":"Creative Common CC-BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Object Tracking","url":"/task/object-tracking","datasets_with_task":"/datasets/task/object-tracking"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Point Tracking","url":"/task/point-tracking","datasets_with_task":"/datasets/task/point-tracking"}],"languages":[],"variants":["Perception Test"],"data_loaders":[],"num_papers_in_archive":10,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-perception-test","task":"Video Question Answering","dataset_variant":"Perception Test","rows":6,"metrics":["Accuracy (Top-1)"],"first_row_in_archive_order":{"model":"Oyrx (34B)","paper":"/paper/oryx-mllm-on-demand-spatial-temporal","metrics":{"Accuracy (Top-1)":"71.4"},"code_links":[{"title":"oryx-mllm/oryx","url":"https://github.com/oryx-mllm/oryx"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-tracking-on-perception-test","task":"Object Tracking","dataset_variant":"Perception Test","rows":1,"metrics":["Average IOU"],"first_row_in_archive_order":{"model":"Siam-FC","paper":"/paper/perception-test-a-diagnostic-benchmark-for-2","metrics":{"Average IOU":"0.66"},"code_links":[{"title":"deepmind/perception_test","url":"https://github.com/deepmind/perception_test"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/point-tracking-on-perception-test","task":"Point Tracking","dataset_variant":"Perception Test","rows":1,"metrics":["Average Jaccard"],"first_row_in_archive_order":{"model":"Static Baseline","paper":"/paper/perception-test-a-diagnostic-benchmark-for-2","metrics":{"Average Jaccard":"0.36"},"code_links":[{"title":"deepmind/perception_test","url":"https://github.com/deepmind/perception_test"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/bimba-selective-scan-compression-for-long","title":"BIMBA: Selective-Scan Compression for Long-Range Video Question Answering","date":"2025-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/oryx-mllm-on-demand-spatial-temporal","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","date":"2024-09-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/traveler-a-multi-lmm-agent-framework-for","title":"TraveLER: A Modular Multi-LMM Agent Framework for Video Question-Answering","date":"2024-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/perception-test-a-diagnostic-benchmark-for-2","title":"Perception Test: A Diagnostic Benchmark for Multimodal Video Models","date":"2023-05-23","rows_on_this_dataset":3,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":23,"samples_ran":10,"samples_unverified":13,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}