{"url":"/dataset/newsroom","name":"NEWSROOM","full_name":"CORNELL NEWSROOM","description_markdown":"CORNELL NEWSROOM is a large dataset for training and evaluating summarization systems. It contains 1.3 million articles and summaries written by authors and editors in the newsrooms of 38 major publications. The summaries are obtained from search and social metadata between 1998 and 2017 and use a variety of summarization strategies combining extraction and abstraction.\r\n\r\nSource: [CORNELL NEWSROOM](http://lil.nlp.cornell.edu/newsroom/)","description_withheld":null,"homepage":"http://lil.nlp.cornell.edu/newsroom/","introduced_date":"2018-04-30","introduced_date_note":null,"introduced_by":{"paper":"/paper/newsroom-a-dataset-of-13-million-summaries","title":"Newsroom: A Dataset of 1.3 Million Summaries with Diverse Extractive Strategies","first_author":"Max Grusky","url":null},"license":{"name":"Custom (non-commercial)","url":"https://cornell.qualtrics.com/jfe/form/SV_6YA3HQ2p75XH4IR"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Abstractive Text Summarization","url":"/task/abstractive-text-summarization","datasets_with_task":"/datasets/task/abstractive-text-summarization"},{"name":"Document Summarization","url":"/task/document-summarization","datasets_with_task":"/datasets/task/document-summarization"}],"languages":[],"variants":["NEWSROOM"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/lil-lab/newsroom","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/newsroom","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/newsroom","frameworks":["tf","jax"]}],"num_papers_in_archive":107,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}