{"url":"/dataset/wikitext-2","name":"WikiText-2","full_name":"WikiText-2","description_markdown":"The WikiText language modeling dataset is a collection of over 100 million tokens extracted from the set of verified Good and Featured articles on Wikipedia. The dataset is available under the Creative Commons Attribution-ShareAlike License.\r\n\r\nCompared to the preprocessed version of Penn Treebank (PTB), WikiText-2 is over 2 times larger and WikiText-103 is over 110 times larger. The WikiText dataset also features a far larger vocabulary and retains the original case, punctuation and numbers - all of which are removed in PTB. As it is composed of full articles, the dataset is well suited for models that can take advantage of long term dependencies.\r\n\r\nSource: [The WikiText Long Term Dependency Language Modeling Dataset](https://blog.einstein.ai/the-wikitext-long-term-dependency-language-modeling-dataset/)\r\nImage Source: [https://blog.einstein.ai/the-wikitext-long-term-dependency-language-modeling-dataset/](https://blog.einstein.ai/the-wikitext-long-term-dependency-language-modeling-dataset/)","description_withheld":null,"homepage":"https://blog.einstein.ai/the-wikitext-long-term-dependency-language-modeling-dataset/","introduced_date":"2016-09-26","introduced_date_note":null,"introduced_by":{"paper":"/paper/pointer-sentinel-mixture-models","title":"Pointer Sentinel Mixture Models","first_author":"Stephen Merity","url":null},"license":{"name":"CC BY-SA 3.0","url":"https://blog.einstein.ai/the-wikitext-long-term-dependency-language-modeling-dataset/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Spanish","url":"/datasets/language/spanish"},{"name":"German","url":"/datasets/language/german"},{"name":"Swedish","url":"/datasets/language/swedish"}],"variants":["WikiText-103","WikiText-2","wikitext wikitext-2-raw-v1"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Salesforce/wikitext","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wikitext","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Nart/abkhaz","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/rajeshradhakrishnan/malayalam_wiki","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/meliascosta/wiki_academic_subjects","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/bfattori/wikitext_document_level","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/alexkueck/tis","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/mindchain/wikitext2","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/pytorch/text","url":"https://pytorch.org/text/stable/datasets.html#torchtext.datasets.WikiText2","frameworks":["pytorch"]}],"num_papers_in_archive":1081,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-wikitext-2","task":"Language Modelling","dataset_variant":"WikiText-2","rows":38,"metrics":["Test perplexity","Validation perplexity","Number of params"],"first_row_in_archive_order":{"model":"SparseGPT (175B, 50% Sparsity)","paper":"/paper/massive-language-models-can-be-accurately","metrics":{"Test perplexity":"8.21"},"code_links":[{"title":"nvidia/tensorrt-model-optimizer","url":"https://github.com/nvidia/tensorrt-model-optimizer"},{"title":"ist-daslab/sparsegpt","url":"https://github.com/ist-daslab/sparsegpt"},{"title":"nvlabs/maskllm","url":"https://github.com/nvlabs/maskllm"},{"title":"baithebest/adagp","url":"https://github.com/baithebest/adagp"},{"title":"baithebest/sparsellm","url":"https://github.com/baithebest/sparsellm"},{"title":"eth-easl/deltazip","url":"https://github.com/eth-easl/deltazip"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/advancing-state-of-the-art-in-language","title":"Advancing State of the Art in Language Modeling","date":"2023-11-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/massive-language-models-can-be-accurately","title":"SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot","date":"2023-01-02","rows_on_this_dataset":5,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":2,"samples_unverified":10,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/egru-event-based-gru-for-activity-sparse","title":"Efficient recurrent architectures through activity sparsity and sparse back-propagation through time","date":"2022-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hydra-a-system-for-large-multi-model-deep","title":"Hydra: A System for Large Multi-Model Deep Learning","date":"2021-10-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-associative-inference-using-fast-1","title":"Learning Associative Inference Using Fast Weight Memory","date":"2020-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/alleviating-sequence-information-loss-with","title":"Alleviating Sequence Information Loss with Data Overlapping and Prime Batch Sizes","date":"2019-09-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mogrifier-lstm","title":"Mogrifier LSTM","date":"2019-09-04","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/improving-neural-language-modeling-via","title":"Improving Neural Language Modeling via Adversarial Training","date":"2019-06-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-residual-output-layers-for-neural","title":"Deep Residual Output Layers for Neural Language Generation","date":"2019-05-14","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/190409408","title":"Language Models with Transformers","date":"2019-04-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/partially-shuffling-the-training-data-to-1","title":"Partially Shuffling the Training Data to Improve Language Models","date":"2019-03-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","rows_on_this_dataset":4,"code_links":21,"syntology":null},{"paper":"/paper/frage-frequency-agnostic-word-representation","title":"FRAGE: Frequency-Agnostic Word Representation","date":"2018-09-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/direct-output-connection-for-a-high-rank","title":"Direct Output Connection for a High-Rank Language Model","date":"2018-08-30","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/improved-language-modeling-by-decoding-the","title":"Improved Language Modeling by Decoding the Past","date":"2018-08-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/breaking-the-softmax-bottleneck-a-high-rank","title":"Breaking the Softmax Bottleneck: A High-Rank RNN Language Model","date":"2017-11-10","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":1,"samples_unverified":22,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fraternal-dropout","title":"Fraternal Dropout","date":"2017-10-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dynamic-evaluation-of-neural-sequence-models","title":"Dynamic Evaluation of Neural Sequence Models","date":"2017-09-21","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gradual-learning-of-recurrent-neural-networks","title":"Gradual Learning of Recurrent Neural Networks","date":"2017-08-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/regularizing-and-optimizing-lstm-language","title":"Regularizing and Optimizing LSTM Language Models","date":"2017-08-07","rows_on_this_dataset":2,"code_links":45,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-the-state-of-the-art-of-evaluation-in","title":"On the State of the Art of Evaluation in Neural Language Models","date":"2017-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-neural-language-models-with-a","title":"Improving Neural Language Models with a Continuous Cache","date":"2016-12-13","rows_on_this_dataset":2,"code_links":14,"syntology":null},{"paper":"/paper/tying-word-vectors-and-word-classifiers-a","title":"Tying Word Vectors and Word Classifiers: A Loss Framework for Language Modeling","date":"2016-11-04","rows_on_this_dataset":2,"code_links":5,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":48,"samples_ran":13,"samples_unverified":35,"pointer_only_for_licence":18,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}