{"url":"/dataset/billion-word-benchmark","name":"Billion Word Benchmark","full_name":null,"description_markdown":"The **One Billion Word** dataset is a dataset for language modeling. The training/held-out data was produced from the WMT 2011 News Crawl data using a combination of Bash shell and Perl scripts.","description_withheld":null,"homepage":"https://code.google.com/archive/p/1-billion-word-language-modeling-benchmark/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/one-billion-word-benchmark-for-measuring","title":"One Billion Word Benchmark for Measuring Progress in Statistical Language Modeling","first_author":"Ciprian Chelba","url":null},"license":{"name":"Apache-2","url":"https://github.com/ciprian-chelba/1-billion-word-language-modeling-benchmark"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Word Embeddings","url":"/task/word-embeddings","datasets_with_task":"/datasets/task/word-embeddings"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["One Billion Word","Billion Word Benchmark"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/billion-word-benchmark/lm1b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/lm1b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/lm1b","frameworks":["tf","jax"]}],"num_papers_in_archive":141,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-one-billion-word","task":"Language Modelling","dataset_variant":"One Billion Word","rows":27,"metrics":["PPL","Number of params","Validation perplexity"],"first_row_in_archive_order":{"model":"MDLM (AR baseline)","paper":"/paper/simple-and-effective-masked-diffusion","metrics":{"Number of params":"110M","PPL":"20.09"},"code_links":[{"title":"kuleshov-group/mdlm","url":"https://github.com/kuleshov-group/mdlm"},{"title":"masa-ue/svdd","url":"https://github.com/masa-ue/svdd"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-one-billion-word","task":"Text Generation","dataset_variant":"One Billion Word","rows":1,"metrics":["JS-4"],"first_row_in_archive_order":{"model":"WGANGP + DGflow","paper":"/paper/refining-deep-generative-models-via-1","metrics":{"JS-4":"0.186"},"code_links":[{"title":"clear-nus/DGflow","url":"https://github.com/clear-nus/DGflow"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/simple-and-effective-masked-diffusion","title":"Simple and Effective Masked Diffusion Language Models","date":"2024-06-11","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/h-transformer-1d-fast-one-dimensional","title":"H-Transformer-1D: Fast One-Dimensional Hierarchical Attention for Sequences","date":"2021-07-25","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":5,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omninet-omnidirectional-representations-from","title":"OmniNet: Omnidirectional Representations from Transformers","date":"2021-03-01","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/when-attention-meets-fast-recurrence-training","title":"When Attention Meets Fast Recurrence: Training Language Models with Reduced Compute","date":"2021-02-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/refining-deep-generative-models-via-1","title":"Refining Deep Generative Models via Discriminator Gradient Flow","date":"2020-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","rows_on_this_dataset":1,"code_links":21,"syntology":null},{"paper":"/paper/the-evolved-transformer","title":"The Evolved Transformer","date":"2019-01-30","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/pay-less-attention-with-lightweight-and","title":"Pay Less Attention with Lightweight and Dynamic Convolutions","date":"2019-01-29","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","rows_on_this_dataset":2,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":143,"samples_ran":63,"samples_unverified":80,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mesh-tensorflow-deep-learning-for","title":"Mesh-TensorFlow: Deep Learning for Supercomputers","date":"2018-11-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adaptive-input-representations-for-neural","title":"Adaptive Input Representations for Neural Language Modeling","date":"2018-09-28","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/factorization-tricks-for-lstm-networks","title":"Factorization tricks for LSTM networks","date":"2017-03-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/outrageously-large-neural-networks-the","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","date":"2017-01-23","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-modeling-with-gated-convolutional","title":"Language Modeling with Gated Convolutional Networks","date":"2016-12-23","rows_on_this_dataset":1,"code_links":11,"syntology":null},{"paper":"/paper/exploring-the-limits-of-language-modeling","title":"Exploring the Limits of Language Modeling","date":"2016-02-07","rows_on_this_dataset":3,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/skip-gram-language-modeling-using-sparse-non","title":"Skip-gram Language Modeling Using Sparse Non-negative Matrix Probability Estimation","date":"2014-12-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/one-billion-word-benchmark-for-measuring","title":"One Billion Word Benchmark for Measuring Progress in Statistical Language Modeling","date":"2013-12-11","rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":169,"samples_ran":75,"samples_unverified":94,"pointer_only_for_licence":50,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}