{"url":"/dataset/booksum","name":"BookSum","full_name":null,"description_markdown":"**BookSum** is a collection of datasets for long-form narrative summarization. This dataset covers source documents from the literature domain, such as novels, plays and stories, and includes highly abstractive, human written summaries on three levels of granularity of increasing difficulty: paragraph-, chapter-, and book-level. The domain and structure of this dataset poses a unique set of challenges for summarization systems, which include: processing very long documents, non-trivial causal and temporal dependencies, and rich discourse structures.\r\n\r\n**BookSum** contains summaries for 142,753 paragraphs, 12,293 chapters and 436 books.","description_withheld":null,"homepage":"https://github.com/salesforce/booksum","introduced_date":"2021-05-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/booksum-a-collection-of-datasets-for-long","title":"BookSum: A Collection of Datasets for Long-form Narrative Summarization","first_author":"Wojciech Kryściński","url":null},"license":{"name":"BSD-3 License","url":"https://github.com/salesforce/booksum/blob/main/LICENSE.txt"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Abstractive Text Summarization","url":"/task/abstractive-text-summarization","datasets_with_task":"/datasets/task/abstractive-text-summarization"},{"name":"Document Summarization","url":"/task/document-summarization","datasets_with_task":"/datasets/task/document-summarization"},{"name":"Long-Form Narrative Summarization","url":"/task/long-form-narrative-summarization","datasets_with_task":"/datasets/task/long-form-narrative-summarization"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["BookSum"],"data_loaders":[{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/booksum","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/salesforce/booksum","url":"https://github.com/salesforce/booksum","frameworks":[]}],"num_papers_in_archive":39,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/long-form-narrative-summarization-on-booksum","task":"Long-Form Narrative Summarization","dataset_variant":"BookSum","rows":10,"metrics":["BERTScore (F1)","ROUGE-1","ROUGE-2","ROUGE-L","ROUGE (geometric mean of 1/2/L)"],"first_row_in_archive_order":{"model":"NexusSum (Mistral Large)","paper":"/paper/nexussum-hierarchical-llm-agents-for-long","metrics":{"BERTScore (F1)":"70.70","ROUGE (geometric mean of 1/2/L)":"18.27","ROUGE-1":"42.51","ROUGE-2":"10.27","ROUGE-L":"23.91"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-summarization-on-booksum","task":"Text Summarization","dataset_variant":"BookSum","rows":3,"metrics":["ROUGE","ROUGE-2","ROUGE-L"],"first_row_in_archive_order":{"model":"Echoes-Extractive-Abstractive","paper":"/paper/echoes-from-alexandria-a-large-resource-for","metrics":{"ROUGE":"42.13","ROUGE-2":"10.53","ROUGE-L":"16.75"},"code_links":[{"title":"babelscape/echoes-from-alexandria","url":"https://github.com/babelscape/echoes-from-alexandria"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/nexussum-hierarchical-llm-agents-for-long","title":"NexusSum: Hierarchical LLM Agents for Long-Form Narrative Summarization","date":"2025-05-30","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/end-to-end-long-document-summarization-using","title":"End-to-End Long Document Summarization using Gradient Caching","date":"2025-01-03","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/chain-of-agents-large-language-models","title":"Chain of Agents: Large Language Models Collaborating on Long-Context Tasks","date":"2024-06-04","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/echoes-from-alexandria-a-large-resource-for","title":"Echoes from Alexandria: A Large Resource for Multilingual Book Summarization","date":"2023-06-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adapting-pretrained-text-to-text-models-for","title":"Adapting Pretrained Text-to-Text Models for Long Text Sequences","date":"2022-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/long-document-summarization-with-top-down-and-1","title":"Long Document Summarization with Top-down and Bottom-up Inference","date":"2022-03-15","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}