{"url":"/dataset/gem","name":"GEM","full_name":"Generation, Evaluation, and Metrics","description_markdown":"Generation, Evaluation, and Metrics (GEM) is a benchmark environment for Natural Language Generation with a focus on its Evaluation, both through human annotations and automated Metrics.\r\n\r\nGEM aims to:\r\n\r\n- measure NLG progress across 13 datasets spanning many NLG tasks and languages.\r\n- provide an in-depth analysis of data and models presented via data statements and challenge sets.\r\n- develop standards for evaluation of generated text using both automated and human metrics.\r\n\r\nIt is our goal to regularly update GEM and to encourage toward more inclusive practices in dataset development by extending existing data or developing datasets for additional languages.\r\n\r\nSource: [https://gem-benchmark.com/](https://gem-benchmark.com/)\r\nImage Source: [Gehrmann et al](https://arxiv.org/pdf/2102.01672.pdf)","description_withheld":null,"homepage":"https://gem-benchmark.com/","introduced_date":"2021-02-02","introduced_date_note":null,"introduced_by":{"paper":"/paper/the-gem-benchmark-natural-language-generation","title":"The GEM Benchmark: Natural Language Generation, its Evaluation and Metrics","first_author":"Sebastian Gehrmann","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Extreme Summarization","url":"/task/extreme-summarization","datasets_with_task":"/datasets/task/extreme-summarization"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["GEM","GEM-XSum"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/GEM/gem","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/gem","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/gem","frameworks":["tf","jax"]}],"num_papers_in_archive":45,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/extreme-summarization-on-gem-xsum","task":"Extreme Summarization","dataset_variant":"GEM-XSum","rows":6,"metrics":["ROUGE-2","BLEU score","Parameters"],"first_row_in_archive_order":{"model":"PEGASUS","paper":"/paper/the-gem-benchmark-natural-language-generation","metrics":{"Parameters":"568 M","ROUGE-2":"23.2"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":3,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":30,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/byt5-towards-a-token-free-future-with-pre","title":"ByT5: Towards a token-free future with pre-trained byte-to-byte models","date":"2021-05-28","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-gem-benchmark-natural-language-generation","title":"The GEM Benchmark: Natural Language Generation, its Evaluation and Metrics","date":"2021-02-02","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":43,"samples_ran":30,"samples_unverified":13,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}