{"url":"/dataset/multi-news","name":"Multi-News","full_name":"Multi-News","description_markdown":"**Multi-News**, consists of news articles and human-written summaries of these articles from the site newser.com. Each summary is professionally written by editors and includes links to the original articles cited.\r\n\r\nSource: [Multi-News: a Large-Scale Multi-Document Summarization Dataset and Abstractive Hierarchical Model](https://arxiv.org/pdf/1906.01749.pdf)\r\nImage Source: [https://arxiv.org/pdf/1906.01749.pdf](https://arxiv.org/pdf/1906.01749.pdf)","description_withheld":null,"homepage":"https://github.com/Alex-Fabbri/Multi-News","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/multi-news-a-large-scale-multi-document","title":"Multi-News: a Large-Scale Multi-Document Summarization Dataset and Abstractive Hierarchical Model","first_author":"Alexander R. Fabbri","url":null},"license":{"name":"Custom","url":"https://github.com/Alex-Fabbri/Multi-News"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Document Summarization","url":"/task/document-summarization","datasets_with_task":"/datasets/task/document-summarization"},{"name":"Summarization","url":"/task/summarization","datasets_with_task":"/datasets/task/summarization"},{"name":"Multi-Document Summarization","url":"/task/multi-document-summarization","datasets_with_task":"/datasets/task/multi-document-summarization"},{"name":"Cross-Document Language Modeling","url":"/task/cross-document-language-modeling","datasets_with_task":"/datasets/task/cross-document-language-modeling"},{"name":"Information Threading","url":"/task/information-threading","datasets_with_task":"/datasets/task/information-threading"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MultiNews val","MultiNews test","Multi-News","multi_news"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/alexfabbri/multi_news","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/multi_news","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multinews_dense_mean","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multinews_dense_oracle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multi_news_sparse","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multinews_sparse_oracle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multinews_sparse_max","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multinews_sparse_mean","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multinews_dense_max","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/multi_news","frameworks":["tf","jax"]},{"repo":"https://github.com/Alex-Fabbri/Multi-News","url":"https://github.com/Alex-Fabbri/Multi-News","frameworks":["pytorch"]}],"num_papers_in_archive":122,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multi-document-summarization-on-multi-news","task":"Multi-Document Summarization","dataset_variant":"Multi-News","rows":6,"metrics":["ROUGE-1","ROUGE-2","ROUGE-SU4","ROUGE-L"],"first_row_in_archive_order":{"model":"PRIMER","paper":"/paper/primer-pyramid-based-masked-sentence-pre","metrics":{"ROUGE-1":"49.9","ROUGE-2":"21.1","ROUGE-L":"25.9"},"code_links":[{"title":"allenai/primer","url":"https://github.com/allenai/primer"},{"title":"allenai/open-mds","url":"https://github.com/allenai/open-mds"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/longformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-document-language-modeling-on-multinews","task":"Cross-Document Language Modeling","dataset_variant":"MultiNews val","rows":3,"metrics":["Perplexity"],"first_row_in_archive_order":{"model":"CD-LM","paper":"/paper/cross-document-language-modeling","metrics":{"Perplexity":"1.69"},"code_links":[{"title":"aviclu/cdlm","url":"https://github.com/aviclu/cdlm"},{"title":"aviclu/CD-LM","url":"https://github.com/aviclu/CD-LM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-document-language-modeling-on-multinews-1","task":"Cross-Document Language Modeling","dataset_variant":"MultiNews test","rows":3,"metrics":["Perplexity"],"first_row_in_archive_order":{"model":"CD-LM","paper":"/paper/cross-document-language-modeling","metrics":{"Perplexity":"1.76"},"code_links":[{"title":"aviclu/cdlm","url":"https://github.com/aviclu/cdlm"},{"title":"aviclu/CD-LM","url":"https://github.com/aviclu/CD-LM"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/information-threading-on-multi-news","task":"Information Threading","dataset_variant":"Multi-News","rows":1,"metrics":["NMI"],"first_row_in_archive_order":{"model":"SeqINT","paper":"/paper/identifying-chronological-and-coherent","metrics":{"NMI":"0.8008"},"code_links":[{"title":"hitt08/HINT","url":"https://github.com/hitt08/HINT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/summarization-on-multi-news","task":"Summarization","dataset_variant":"multi_news","rows":0,"metrics":["ROUGE-1","ROUGE-2","ROUGE-L","ROUGE-LSUM","gen_len","loss"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/identifying-chronological-and-coherent","title":"Identifying chronological and coherent information threads using 5W1H questions and temporal relationships","date":"2023-01-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/longt5-efficient-text-to-text-transformer-for","title":"LongT5: Efficient Text-To-Text Transformer for Long Sequences","date":"2021-12-15","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/primer-pyramid-based-masked-sentence-pre","title":"PRIMERA: Pyramid-based Masked Sentence Pre-training for Multi-document Summarization","date":"2021-10-16","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-document-summarization","title":"Multi-Document Summarization withDeterminantal Point Process Attention","date":"2021-07-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cross-document-language-modeling","title":"CDLM: Cross-Document Language Modeling","date":"2021-01-02","rows_on_this_dataset":6,"code_links":2,"syntology":null},{"paper":"/paper/multi-news-a-large-scale-multi-document","title":"Multi-News: a Large-Scale Multi-Document Summarization Dataset and Abstractive Hierarchical Model","date":"2019-06-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bottom-up-abstractive-summarization","title":"Bottom-Up Abstractive Summarization","date":"2018-08-31","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":14,"samples_ran":10,"samples_unverified":4,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}