{"url":"/dataset/ccnet","name":"CCNet","full_name":"CCNet","description_markdown":"CCNet is a dataset extracted from Common Crawl with a different filtering process than for OSCAR. It was built using a language model trained on Wikipedia, in order to filter out bad quality texts such as code or tables. CCNet contains longer documents on average compared to OSCAR with smaller—and often noisier—documents weeded out.\r\n\r\nSource: [Martin et al](https://arxiv.org/pdf/1911.03894.pdf)","description_withheld":null,"homepage":"https://github.com/facebookresearch/cc_net","introduced_date":"2019-11-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/ccnet-extracting-high-quality-monolingual","title":"CCNet: Extracting High Quality Monolingual Datasets from Web Crawl Data","first_author":"Guillaume Wenzek","url":null},"license":{"name":"MIT (code only)","url":"https://github.com/facebookresearch/cc_net"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"German","url":"/datasets/language/german"}],"variants":["CCNet"],"data_loaders":[{"repo":"https://github.com/facebookresearch/cc_net","url":"https://github.com/facebookresearch/cc_net","frameworks":[]}],"num_papers_in_archive":58,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}