{"url":"/dataset/bigpatent","name":"BigPatent","full_name":null,"description_markdown":"Consists of 1.3 million records of U.S. patent documents along with human written abstractive summaries.\r\n\r\nSource: [BIGPATENT: A Large-Scale Dataset for Abstractive and Coherent Summarization](https://arxiv.org/pdf/1906.03741v1.pdf)","description_withheld":null,"homepage":"https://evasharma.github.io/bigpatent/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/bigpatent-a-large-scale-dataset-for","title":"BIGPATENT: A Large-Scale Dataset for Abstractive and Coherent Summarization","first_author":"Eva Sharma","url":null},"license":{"name":"Unknown","url":null},"modalities":[],"tasks":[{"name":"Text Summarization","url":"/task/text-summarization","datasets_with_task":"/datasets/task/text-summarization"},{"name":"Summarization","url":"/task/summarization","datasets_with_task":"/datasets/task/summarization"}],"languages":[],"variants":["BigPatent","big_patent"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/big_patent","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Trelis/big_patent_sample","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Trelis/big_patent_100k_characters","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/NortheasternUniversity/big_patent","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/big_patent","frameworks":["tf","jax"]}],"num_papers_in_archive":50,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-summarization-on-bigpatent","task":"Text Summarization","dataset_variant":"BigPatent","rows":2,"metrics":["ROUGE-1","ROUGE-2","ROUGE-L"],"first_row_in_archive_order":{"model":"LongT5","paper":"/paper/longt5-efficient-text-to-text-transformer-for","metrics":{"ROUGE-1":"76.87","ROUGE-2":"66.06","ROUGE-L":"70.76"},"code_links":[{"title":"google-research/longt5","url":"https://github.com/google-research/longt5"},{"title":"utsjiyaoli/qa-attack","url":"https://github.com/utsjiyaoli/qa-attack"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/longt5"},{"title":"MindSpore-scientific-2/code-14","url":"https://github.com/MindSpore-scientific-2/code-14/tree/main/longt5"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/summarization-on-big-patent","task":"Summarization","dataset_variant":"big_patent","rows":0,"metrics":["ROUGE-1","ROUGE-2","ROUGE-L","ROUGE-LSUM","gen_len","loss"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/longt5-efficient-text-to-text-transformer-for","title":"LongT5: Efficient Text-To-Text Transformer for Long Sequences","date":"2021-12-15","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":16,"samples_ran":11,"samples_unverified":5,"pointer_only_for_licence":12,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}