{"url":"/dataset/govreport","name":"GovReport","full_name":null,"description_markdown":"GovReport is a dataset for long document summarization, with significantly longer documents and summaries. It consists of reports written by government research agencies including Congressional Research Service and U.S. Government Accountability Office.\r\n\r\nCompared with other long document summarization datasets, government report dataset has longer summaries and documents and requires reading in more context to cover salient words to be summarized.","description_withheld":null,"homepage":"https://gov-report-data.github.io/","introduced_date":"2021-04-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/efficient-attentions-for-long-document","title":"Efficient Attentions for Long Document Summarization","first_author":"Luyang Huang","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Extractive Text Summarization","url":"/task/extractive-document-summarization","datasets_with_task":"/datasets/task/extractive-document-summarization"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["GovReport"],"data_loaders":[{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/gov_report","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":83,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/extractive-text-summarization-on-govreport","task":"Extractive Text Summarization","dataset_variant":"GovReport","rows":2,"metrics":["Avg. Test Rouge1","Avg. Test Rouge2","Avg. Test RougeLsum"],"first_row_in_archive_order":{"model":"MemSum (extractive)","paper":"/paper/memsum-extractive-summarization-of-long","metrics":{"Avg. Test Rouge1":"59.43","Avg. Test Rouge2":"28.60","Avg. Test RougeLsum":"56.69"},"code_links":[{"title":"nianlonggu/memsum","url":"https://github.com/nianlonggu/memsum"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-summarization-on-govreport","task":"Text Summarization","dataset_variant":"GovReport","rows":2,"metrics":["ROUGE-1","ROUGE-2","ROUGE-L"],"first_row_in_archive_order":{"model":"FactorSum","paper":"/paper/factorizing-content-and-budget-decisions-in","metrics":{"ROUGE-1":"60.1","ROUGE-2":"25.28","ROUGE-L":"56.65"},"code_links":[{"title":"thefonseca/factorsum","url":"https://github.com/thefonseca/factorsum"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/adapting-pretrained-text-to-text-models-for","title":"Adapting Pretrained Text-to-Text Models for Long Text Sequences","date":"2022-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/factorizing-content-and-budget-decisions-in","title":"Factorizing Content and Budget Decisions in Abstractive Summarization of Long Documents","date":"2022-05-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/memsum-extractive-summarization-of-long","title":"MemSum: Extractive Summarization of Long Documents Using Multi-Step Episodic Markov Decision Processes","date":"2021-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/factorized-attention-self-attention-with","title":"Efficient Attention: Attention with Linear Complexities","date":"2018-12-04","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}