{"url":"/dataset/wikisum","name":"WikiSum","full_name":"WikiSum","description_markdown":"**WikiSum** is a dataset based on English Wikipedia and suitable for a task of multi-document abstractive summarization. In each instance, the input is comprised of a Wikipedia topic (title of article) and a collection of non-Wikipedia reference documents, and the target is the Wikipedia article text. The dataset is restricted to the articles with at least one crawlable citation. The official split divides the articles roughly into 80/10/10 for train/development/test subsets, resulting in 1865750, 233252, and 232998 examples respectively.\r\n\r\nSource: [Generating Wikipedia by Summarizing Long Sequences](https://arxiv.org/pdf/1801.10198.pdf)\r\nImage Source: [https://arxiv.org/pdf/1801.10198.pdf](https://arxiv.org/pdf/1801.10198.pdf)","description_withheld":null,"homepage":"https://github.com/tensorflow/tensor2tensor/tree/master/tensor2tensor/data_generators/wikisum","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/generating-wikipedia-by-summarizing-long","title":"Generating Wikipedia by Summarizing Long Sequences","first_author":"Peter J. Liu","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Abstractive Text Summarization","url":"/task/abstractive-text-summarization","datasets_with_task":"/datasets/task/abstractive-text-summarization"},{"name":"Document Summarization","url":"/task/document-summarization","datasets_with_task":"/datasets/task/document-summarization"},{"name":"Multi-Document Summarization","url":"/task/multi-document-summarization","datasets_with_task":"/datasets/task/multi-document-summarization"}],"languages":[],"variants":["WikiSum"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/d0rj/wikisum","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/tensor2tensor","url":"https://github.com/tensorflow/tensor2tensor","frameworks":["tf"]}],"num_papers_in_archive":54,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}