{"url":"/dataset/scrolls","name":"SCROLLS","full_name":"Standardized CompaRison Over Long Language Sequences","description_markdown":"** SCROLLS** (Standardized CompaRison Over Long Language Sequences) is an NLP benchmark consisting of a suite of tasks that require **reasoning over long texts**. SCROLLS contains summarization, question answering, and natural language inference tasks, covering multiple domains, including literature, science, business, and entertainment. The dataset is made available in a unified text-to-text format and host a live leaderboard to facilitate research on model architecture and pretraining methods.\r\n\r\nThe **SCROLLS** benchmark contains the datasets [GovReport](govreport), SummScreenFD, [QMSum](qmsum), [QASPER](qasper), [NarrativeQA](NarrativeQA), QuALITY and ContractNLI.","description_withheld":null,"homepage":"https://www.scrolls-benchmark.com","introduced_date":"2022-01-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/scrolls-standardized-comparison-over-long","title":"SCROLLS: Standardized CompaRison Over Long Language Sequences","first_author":"Uri Shaham","url":null},"license":{"name":"MIT","url":"https://github.com/tau-nlp/scrolls/blob/main/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"},{"name":"Long-range modeling","url":"/task/long-range-modeling","datasets_with_task":"/datasets/task/long-range-modeling"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SCROLLS"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/tau/scrolls","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/scrolls","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":42,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/long-range-modeling-on-scrolls","task":"Long-range modeling","dataset_variant":"SCROLLS","rows":13,"metrics":["Avg.","GovRep","SumScr","QMSum","Qspr","Nrtv","QALT EM-T/H","CNLI"],"first_row_in_archive_order":{"model":"CoLT5 XL","paper":"/paper/colt5-faster-long-range-transformers-with","metrics":{"Avg.":"43.51","CNLI":"88.4","GovRep":"61.3/32.2/33.8","Nrtv":"31.1","QALT EM-T/H":"48.1/43.8","QMSum":"36.2/12.9/24.3","Qspr":"53.9","SumScr":"36.4/10.2/21.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/colt5-faster-long-range-transformers-with","title":"CoLT5: Faster Long-Range Transformers with Conditional Computation","date":"2023-03-17","rows_on_this_dataset":1,"code_links":0,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adapting-pretrained-text-to-text-models-for","title":"Adapting Pretrained Text-to-Text Models for Long Text Sequences","date":"2022-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/investigating-efficiently-extending","title":"Investigating Efficiently Extending Transformers for Long Input Summarization","date":"2022-08-08","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/efficient-long-text-understanding-with-short","title":"Efficient Long-Text Understanding with Short-Text Models","date":"2022-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":0,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scrolls-standardized-comparison-over-long","title":"SCROLLS: Standardized CompaRison Over Long Language Sequences","date":"2022-01-10","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/longt5-efficient-text-to-text-transformer-for","title":"LongT5: Efficient Text-To-Text Transformer for Long Sequences","date":"2021-12-15","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":34,"samples_ran":3,"samples_unverified":31,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}