{"url":"/dataset/platinumbench","name":"PlatinumBench","full_name":null,"description_markdown":"Platinum Benchmarks are benchmarks that are are carefully curated to minimize label errors and ambiguity, allowing us to measure reliability of models.\r\n\r\nThis dataset contains fifteen platinum benchmarks created by manually revising questions from existing datasets (see the github repo for details on accessing our revised subset of VQA). To revise each benchmark, we ran a variety of frontier models on individual examples and manually re-annotated any example for which at least one model made an error. See the paper for further details on the revision process.","description_withheld":null,"homepage":"http://platinum-bench.csail.mit.edu/","introduced_date":"2025-02-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/do-large-language-model-benchmarks-test","title":"Do Large Language Model Benchmarks Test Reliability?","first_author":"Joshua Vendrow","url":null},"license":{"name":"CC-BY-SA-4.0","url":null},"modalities":[],"tasks":[],"languages":[],"variants":["PlatinumBench"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}