{"url":"/dataset/gazeta","name":"Gazeta","full_name":null,"description_markdown":"**Gazeta** is a dataset for automatic summarization of Russian news. The dataset consists of 63,435 text-summary pairs. To form training, validation, and test datasets, these pairs were sorted by time and the first 52,400 pairs are used as the training dataset, the proceeding 5,265 pairs as the validation dataset, and the remaining 5,770 pairs as the test dataset.\n\nSource: [https://github.com/IlyaGusev/gazeta](https://github.com/IlyaGusev/gazeta)","description_withheld":null,"homepage":"https://github.com/IlyaGusev/gazeta","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/dataset-for-automatic-summarization-of","title":"Dataset for Automatic Summarization of Russian News","first_author":"Ilya Gusev","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"}],"languages":[{"name":"Russian","url":"/datasets/language/russian"}],"variants":["Gazeta"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/IlyaGusev/gazeta","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/IlyaGusev/gazeta","url":"https://github.com/IlyaGusev/gazeta","frameworks":[]}],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-summarization-on-gazeta","task":"Text Summarization","dataset_variant":"Gazeta","rows":1,"metrics":["ROUGE-1","ROUGE-2","ROUGE-L","BLEU","Meteor"],"first_row_in_archive_order":{"model":"Finetuned mBART","paper":"/paper/dataset-for-automatic-summarization-of","metrics":{"BLEU":"12.4","Meteor":"25.7","ROUGE-1":"32.1","ROUGE-2":"14.2","ROUGE-L":"27.9"},"code_links":[{"title":"IlyaGusev/summarus","url":"https://github.com/IlyaGusev/summarus"},{"title":"IlyaGusev/gazeta","url":"https://github.com/IlyaGusev/gazeta"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/dataset-for-automatic-summarization-of","title":"Dataset for Automatic Summarization of Russian News","date":"2020-06-19","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}