{"url":"/dataset/hutter-prize","name":"Hutter Prize","full_name":null,"description_markdown":"The Hutter Prize Wikipedia dataset, also known as enwiki8, is a byte-level dataset consisting of the first 100 million bytes of a Wikipedia XML dump. For simplicity we shall refer to it as a character-level dataset. Within these 100 million bytes are 205 unique tokens.\r\n\r\nSource: [NLP Progress](http://nlpprogress.com/english/language_modeling.html)","description_withheld":null,"homepage":"http://prize.hutter1.net/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Hutter Prize"],"data_loaders":[],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-hutter-prize","task":"Language Modelling","dataset_variant":"Hutter Prize","rows":18,"metrics":["Bit per Character (BPC)","Number of params"],"first_row_in_archive_order":{"model":"Transformer-XL + RMS dynamic eval","paper":"/paper/dynamic-evaluation-of-transformer-language","metrics":{"Bit per Character (BPC)":"0.94","Number of params":"277M"},"code_links":[{"title":"benkrause/dynamiceval-transformer","url":"https://github.com/benkrause/dynamiceval-transformer"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/longformer-the-long-document-transformer","title":"Longformer: The Long-Document Transformer","date":"2020-04-10","rows_on_this_dataset":2,"code_links":22,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":35,"samples_ran":15,"samples_unverified":20,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/compressive-transformers-for-long-range-1","title":"Compressive Transformers for Long-Range Sequence Modelling","date":"2019-11-13","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mogrifier-lstm","title":"Mogrifier LSTM","date":"2019-09-04","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/dynamic-evaluation-of-transformer-language","title":"Dynamic Evaluation of Transformer Language Models","date":"2019-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","rows_on_this_dataset":3,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":143,"samples_ran":63,"samples_unverified":80,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/character-level-language-modeling-with-deeper","title":"Character-Level Language Modeling with Deeper Self-Attention","date":"2018-08-09","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/an-analysis-of-neural-language-modeling-at","title":"An Analysis of Neural Language Modeling at Multiple Scales","date":"2018-03-22","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":3,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dynamic-evaluation-of-neural-sequence-models","title":"Dynamic Evaluation of Neural Sequence Models","date":"2017-09-21","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fast-slow-recurrent-neural-networks","title":"Fast-Slow Recurrent Neural Networks","date":"2017-05-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multiplicative-lstm-for-sequence-modelling","title":"Multiplicative LSTM for sequence modelling","date":"2016-09-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/recurrent-highway-networks","title":"Recurrent Highway Networks","date":"2016-07-12","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":213,"samples_ran":85,"samples_unverified":128,"pointer_only_for_licence":48,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}