{"url":"/dataset/thucnews","name":"THUCNews","full_name":"THU Chinese Text Classification","description_markdown":"The THUCNews Chinese text dataset is a large-scale Chinese text classification dataset. It contains approximately **840,000 news documents** categorized into **14 classes**. The dataset was generated by filtering historical data from the Sina News RSS feeds between 2005 and 2011. This dataset can be used for various tasks such as text classification and training word vectors.","description_withheld":null,"homepage":"https://github.com/laomagic/THUCNewsProject","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Chinese Document Classification","url":"/task/chinese-document-classification","datasets_with_task":"/datasets/task/chinese-document-classification"}],"languages":[],"variants":["THUCNews"],"data_loaders":[{"repo":"https://github.com/Graviti-AI/datasets","url":"https://gas.graviti.com/dataset/graviti/THUCNews","frameworks":["tf","pytorch"]}],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}