{"url":"/dataset/text8","name":"Text8","full_name":null,"description_markdown":"Desc: [About of Text8](http://mattmahoney.net/dc/textdata.html)","description_withheld":null,"homepage":"http://mattmahoney.net/dc/textdata.html","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Text8"],"data_loaders":[],"num_papers_in_archive":22,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-text8","task":"Language Modelling","dataset_variant":"Text8","rows":24,"metrics":["Bit per Character (BPC)","Number of params"],"first_row_in_archive_order":{"model":"GPT-2","paper":"/paper/language-models-are-unsupervised-multitask","metrics":{"Bit per Character (BPC)":"0.98","Number of params":"1542M"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"openai/gpt-2","url":"https://github.com/openai/gpt-2"},{"title":"PaddlePaddle/PaddleNLP","url":"https://github.com/PaddlePaddle/PaddleNLP/tree/develop/examples/language_model/gpt"},{"title":"minimaxir/gpt-2-simple","url":"https://github.com/minimaxir/gpt-2-simple"},{"title":"imcaspar/gpt2-ml","url":"https://github.com/imcaspar/gpt2-ml"},{"title":"huggingface/swift-coreml-transformers","url":"https://github.com/huggingface/swift-coreml-transformers"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/blob/master/research/nlp/gpt2"},{"title":"jankrepl/mildlyoverfitted","url":"https://github.com/jankrepl/mildlyoverfitted"},{"title":"affjljoo3581/GPT2","url":"https://github.com/affjljoo3581/GPT2"},{"title":"akanyaani/gpt-2-tensorflow2.0","url":"https://github.com/akanyaani/gpt-2-tensorflow2.0"},{"title":"lvyufeng/bert4ms","url":"https://github.com/lvyufeng/bert4ms"},{"title":"milmor/GPT","url":"https://github.com/milmor/GPT"},{"title":"abhaskumarsinha/MinimalGPT","url":"https://github.com/abhaskumarsinha/MinimalGPT"},{"title":"aananda-giri/gpt2-nepali","url":"https://github.com/aananda-giri/gpt2-nepali"},{"title":"akanyaani/minGPTF","url":"https://github.com/akanyaani/minGPTF"},{"title":"abhaskumarsinha/Corpus2GPT","url":"https://github.com/abhaskumarsinha/Corpus2GPT"},{"title":"VachanVY/gpt.jax","url":"https://github.com/VachanVY/gpt.jax"},{"title":"MS-P3/code5","url":"https://github.com/MS-P3/code5/tree/main/gpt2"},{"title":"ramanakshay/nanogpt","url":"https://github.com/ramanakshay/nanogpt"},{"title":"2023-MindSpore-1/ms-code-154","url":"https://github.com/2023-MindSpore-1/ms-code-154"},{"title":"varun-suresh/experiments-with-gpt2","url":"https://github.com/varun-suresh/experiments-with-gpt2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/bayesian-flow-networks","title":"Bayesian Flow Networks","date":"2023-08-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":11,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/2305-14952","title":"Focus Your Attention (with Adaptive IIR Filters)","date":"2023-05-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/long-short-transformer-efficient-transformers","title":"Long-Short Transformer: Efficient Transformers for Language and Vision","date":"2021-07-05","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pay-attention-when-required","title":"Pay Attention when Required","date":"2020-09-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/recurrent-highway-networks-with-grouped","title":"Recurrent Highway Networks with Grouped Auxiliary Memory","date":"2019-12-13","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/bp-transformer-modelling-long-range-context","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","date":"2019-11-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/augmenting-self-attention-with-persistent","title":"Augmenting Self-attention with Persistent Memory","date":"2019-07-02","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/discrete-flows-invertible-generative-models","title":"Discrete Flows: Invertible Generative Models of Discrete Data","date":"2019-05-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/adaptive-attention-span-in-transformers","title":"Adaptive Attention Span in Transformers","date":"2019-05-19","rows_on_this_dataset":2,"code_links":8,"syntology":null},{"paper":"/paper/dynamic-evaluation-of-transformer-language","title":"Dynamic Evaluation of Transformer Language Models","date":"2019-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","rows_on_this_dataset":1,"code_links":21,"syntology":null},{"paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","rows_on_this_dataset":1,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":143,"samples_ran":63,"samples_unverified":80,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/character-level-language-modeling-with-deeper","title":"Character-Level Language Modeling with Deeper Self-Attention","date":"2018-08-09","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/dynamic-evaluation-of-neural-sequence-models","title":"Dynamic Evaluation of Neural Sequence Models","date":"2017-09-21","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multiplicative-lstm-for-sequence-modelling","title":"Multiplicative LSTM for sequence modelling","date":"2016-09-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-multiscale-recurrent-neural","title":"Hierarchical Multiscale Recurrent Neural Networks","date":"2016-09-06","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/recurrent-highway-networks","title":"Recurrent Highway Networks","date":"2016-07-12","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/recurrent-batch-normalization","title":"Recurrent Batch Normalization","date":"2016-03-30","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/architectural-complexity-measures-of","title":"Architectural Complexity Measures of Recurrent Neural Networks","date":"2016-02-26","rows_on_this_dataset":2,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":8,"samples_harvested":181,"samples_ran":84,"samples_unverified":97,"pointer_only_for_licence":49,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}