{"url":"/dataset/wiki-40b","name":"Wiki-40B","full_name":null,"description_markdown":"A new multilingual language model benchmark that is composed of 40+ languages spanning several scripts and linguistic families containing round 40 billion characters and aimed to accelerate the research of multilingual modeling.\r\n\r\nSource: [Wiki-40B: Multilingual Language Model Dataset](https://www.shaip.com/blog/multimodal-large-language-models-mllms/)","description_withheld":null,"homepage":"https://research.google/pubs/pub49029/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/wiki-40b-multilingual-language-model-dataset","title":"Wiki-40B: Multilingual Language Model Dataset","first_author":"M. Guo","url":null},"license":null,"modalities":[],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Quantization","url":"/task/quantization","datasets_with_task":"/datasets/task/quantization"},{"name":"Benchmarking","url":"/task/benchmarking","datasets_with_task":"/datasets/task/benchmarking"}],"languages":[],"variants":["Wiki-40B"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/google/wiki40b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/wiki40b","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/wiki40b","frameworks":["tf","jax"]}],"num_papers_in_archive":30,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-wiki-40b","task":"Language Modelling","dataset_variant":"Wiki-40B","rows":3,"metrics":["Perplexity"],"first_row_in_archive_order":{"model":"FLASH-Quad-8k","paper":"/paper/transformer-quality-in-linear-time","metrics":{"Perplexity":"14.998"},"code_links":[{"title":"lucidrains/FLASH-pytorch","url":"https://github.com/lucidrains/FLASH-pytorch"},{"title":"zhuiyitechnology/gau-alpha","url":"https://github.com/zhuiyitechnology/gau-alpha"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/benchmarking-on-wiki-40b","task":"Benchmarking","dataset_variant":"Wiki-40B","rows":1,"metrics":["Perplexity"],"first_row_in_archive_order":{"model":"OutEffHop-Bert_base","paper":"/paper/outlier-efficient-hopfield-layers-for-large","metrics":{"Perplexity":"6.209"},"code_links":[{"title":"magics-lab/outeffhop","url":"https://github.com/magics-lab/outeffhop"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/quantization-on-wiki-40b","task":"Quantization","dataset_variant":"Wiki-40B","rows":1,"metrics":["Perplexity"],"first_row_in_archive_order":{"model":"OutEffHop-Bert_base","paper":"/paper/outlier-efficient-hopfield-layers-for-large","metrics":{"Perplexity":"6.295"},"code_links":[{"title":"magics-lab/outeffhop","url":"https://github.com/magics-lab/outeffhop"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/outlier-efficient-hopfield-layers-for-large","title":"Outlier-Efficient Hopfield Layers for Large Transformer-Based Models","date":"2024-04-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":10,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transformer-quality-in-linear-time","title":"Transformer Quality in Linear Time","date":"2022-02-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/combiner-full-attention-transformer-with","title":"Combiner: Full Attention Transformer with Sparse Computation Cost","date":"2021-07-12","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":16,"samples_ran":11,"samples_unverified":5,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}