{"url":"/dataset/c4","name":"C4","full_name":"Colossal Clean Crawled Corpus","description_markdown":"**C4** is a colossal, cleaned version of Common Crawl's web crawl corpus. It was based on Common Crawl dataset: https://commoncrawl.org. It was used to train the T5 text-to-text Transformer models.\r\n\r\nThe dataset can be downloaded in a pre-processed form from [allennlp](https://github.com/allenai/allennlp/discussions/5056).","description_withheld":null,"homepage":"https://github.com/google-research/text-to-text-transfer-transformer#c4","introduced_date":"2019-10-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","first_author":"Colin Raffel","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[{"name":"Thai","url":"/datasets/language/thai"}],"variants":["C4","c4 en"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Peihao/test-dateset","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/c4","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Birchlabs/c4-t5-ragged","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/legacy-datasets/c4","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/c4","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/google-research/text-to-text-transfer-transformer","url":"https://github.com/google-research/text-to-text-transfer-transformer","frameworks":["tf"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/c4","frameworks":["tf","jax"]}],"num_papers_in_archive":981,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-modelling-on-c4","task":"Language Modelling","dataset_variant":"C4","rows":9,"metrics":["Perplexity","TPUv3 Hours","Steps"],"first_row_in_archive_order":{"model":"Primer","paper":"/paper/primer-searching-for-efficient-transformers","metrics":{"Perplexity":"12.35","Steps":"1M","TPUv3 Hours":"17.3K"},"code_links":[{"title":"labmlai/annotated_deep_learning_paper_implementations","url":"https://github.com/labmlai/annotated_deep_learning_paper_implementations"},{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/primer"},{"title":"lucidrains/FLASH-pytorch","url":"https://github.com/lucidrains/FLASH-pytorch"},{"title":"JunnYu/x-transformers-paddle","url":"https://github.com/JunnYu/x-transformers-paddle"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/llm-int8-8-bit-matrix-multiplication-for","title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","date":"2022-08-15","rows_on_this_dataset":4,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/n-grammer-augmenting-transformers-with-latent-1","title":"N-Grammer: Augmenting Transformers with latent n-grams","date":"2022-07-13","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/primer-searching-for-efficient-transformers","title":"Primer: Searching for Efficient Transformers for Language Modeling","date":"2021-09-17","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}