{"url":"/dataset/multi-xscience","name":"Multi-XScience","full_name":null,"description_markdown":"**Multi-XScience** is a large-scale dataset for multi-document summarization of scientific articles. It has 30,369, 5,066 and 5,093 samples for the train, validation and test split respectively. The average document length is 778.08 words and the average summary length is 116.44 words.\n\nSource: [https://github.com/yaolu/Multi-XScience](https://github.com/yaolu/Multi-XScience)","description_withheld":null,"homepage":"https://github.com/yaolu/Multi-XScience","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/multi-xscience-a-large-scale-dataset-for","title":"Multi-XScience: A Large-scale Dataset for Extreme Multi-document Summarization of Scientific Articles","first_author":"Yao Lu","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Document Summarization","url":"/task/document-summarization","datasets_with_task":"/datasets/task/document-summarization"},{"name":"Multi-Document Summarization","url":"/task/multi-document-summarization","datasets_with_task":"/datasets/task/multi-document-summarization"}],"languages":[],"variants":["Multi-XScience"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/yaolu/multi_x_science_sum","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/multi_x_science_sum","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/arka0821/multi_document_summarization_test","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multi_x_science_sum_sparse","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multi_x_science_sum_sparse_oracle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multixscience_sparse_oracle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multixscience_sparse_mean","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multixscience_sparse_max","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multixscience_dense_max","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multixscience_dense_mean","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/multixscience_dense_oracle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/yaolu/Multi-XScience","url":"https://github.com/yaolu/Multi-XScience","frameworks":[]}],"num_papers_in_archive":10,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}