{"url":"/dataset/wcep","name":"WCEP","full_name":"Wikipedia Current Events Portal","description_markdown":"The WCEP dataset for multi-document summarization (MDS) consists of short, human-written summaries about news events, obtained from the Wikipedia Current Events Portal (WCEP), each paired with a cluster of news articles associated with an event. These articles consist of sources cited by editors on WCEP, and are extended with articles automatically obtained from the Common Crawl News dataset. \r\n\r\nSource: [WCEP](https://github.com/complementizer/wcep-mds-dataset)","description_withheld":null,"homepage":"https://github.com/complementizer/wcep-mds-dataset","introduced_date":"2020-05-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-large-scale-multi-document-summarization","title":"A Large-Scale Multi-Document Summarization Dataset from the Wikipedia Current Events Portal","first_author":"Demian Gholipour Ghalandari","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Document Summarization","url":"/task/document-summarization","datasets_with_task":"/datasets/task/document-summarization"},{"name":"Multi-Document Summarization","url":"/task/multi-document-summarization","datasets_with_task":"/datasets/task/multi-document-summarization"}],"languages":[],"variants":["WCEP"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/wcep_sparse_oracle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/wcep_sparse_max","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/wcep_sparse_mean","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/wcep_dense_mean","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/wcep_dense_max","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/wcep_dense_oracle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/complementizer/wcep-mds-dataset","url":"https://github.com/complementizer/wcep-mds-dataset","frameworks":[]}],"num_papers_in_archive":31,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multi-document-summarization-on-wcep","task":"Multi-Document Summarization","dataset_variant":"WCEP","rows":1,"metrics":["ROUGE-1","ROUGE-2","ROUGE-L"],"first_row_in_archive_order":{"model":"PRIMER","paper":"/paper/primer-pyramid-based-masked-sentence-pre","metrics":{"ROUGE-1":"46.1","ROUGE-2":"25.2","ROUGE-L":"37.9"},"code_links":[{"title":"allenai/primer","url":"https://github.com/allenai/primer"},{"title":"allenai/open-mds","url":"https://github.com/allenai/open-mds"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/longformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/primer-pyramid-based-masked-sentence-pre","title":"PRIMERA: Pyramid-based Masked Sentence Pre-training for Multi-document Summarization","date":"2021-10-16","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}