{"url":"/dataset/cci-3-0-hq","name":"CCI 3.0-HQ","full_name":null,"description_markdown":"To address the scarcity of high-quality safety datasets in the Chinese, we open-sourced the CCI (Chinese Corpora Internet) dataset on November 29, 2023. Building on this foundation, we continue to expand the data source, adopt stricter data cleaning methods, and complete the construction of the CCI 3.0 dataset. This dataset is composed of high-quality, reliable Internet data from trusted sources. And then with more stricter filtering, The CCI 3.0 HQ corpus released is about 500GB in size.","description_withheld":null,"homepage":"https://huggingface.co/datasets/BAAI/CCI3-HQ","introduced_date":"2024-10-24","introduced_date_note":null,"introduced_by":{"paper":"/paper/cci3-0-hq-a-large-scale-chinese-dataset-of","title":"CCI3.0-HQ: a large-scale Chinese dataset of high quality designed for pre-training large language models","first_author":"Liangdong Wang","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["CCI 3.0-HQ"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}