{"url":"/dataset/quality","name":"QuALITY","full_name":"Question Answering with Long Input Texts, Yes!","description_markdown":"**QuALITY** (**Question Answering with Long Input Texts, Yes!**) is a multiple-choice question answering dataset for long document comprehension. The dataset consists of context passages in English that have an average length of about 5,000 tokens, much longer than typical current models can process. Unlike in prior work with passages, the questions are written and validated by contributors who have read the entire passage, rather than relying on summaries or excerpts.","description_withheld":null,"homepage":"https://github.com/nyu-mll/quality","introduced_date":"2021-12-16","introduced_date_note":null,"introduced_by":{"paper":"/paper/quality-question-answering-with-long-input","title":"QuALITY: Question Answering with Long Input Texts, Yes!","first_author":"Richard Yuanzhe Pang","url":null},"license":{"name":"CC BY 4.0 License","url":"http://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Long Form Question Answering","url":"/task/long-form-question-answering","datasets_with_task":"/datasets/task/long-form-question-answering"}],"languages":[],"variants":["QuALITY"],"data_loaders":[],"num_papers_in_archive":98,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-quality","task":"Question Answering","dataset_variant":"QuALITY","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Claude 1.3 (5-shot)","paper":"/paper/model-card-and-evaluations-for-claude-models","metrics":{"Accuracy":"84.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/raptor-recursive-abstractive-processing-for","title":"RAPTOR: Recursive Abstractive Processing for Tree-Organized Retrieval","date":"2024-01-31","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","rows_on_this_dataset":3,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}