{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/lsicc-a-large-scale-informal-chinese-corpus","title":"LSICC: A Large Scale Informal Chinese Corpus","arxiv_id":"1811.10167","date":"2018-11-26","proceeding":null,"authors":["Jianyu Zhao","Zhuoran Ji"],"abstract":"Deep learning based natural language processing model is proven powerful, but\nneed large-scale dataset. Due to the significant gap between the real-world\ntasks and existing Chinese corpus, in this paper, we introduce a large-scale\ncorpus of informal Chinese. This corpus contains around 37 million book reviews\nand 50 thousand netizen's comments to the news. We explore the informal words\nfrequencies of the corpus and show the difference between our corpus and the\nexisting ones. The corpus can be further used to train deep learning based\nnatural language processing tasks such as Chinese word segmentation, sentiment\nanalysis.","url_abs":"http://arxiv.org/abs/1811.10167v1","url_pdf":"http://arxiv.org/pdf/1811.10167v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"lsicc-a-large-scale-informal-chinese-corpus","repo_url":"https://github.com/JaniceZhao/Chinese-Forum-Corpus","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"none","reach":null},{"paper_slug":"lsicc-a-large-scale-informal-chinese-corpus","repo_url":"https://github.com/JaniceZhao/Douban-Dushu-Dataset","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"chinese-word-segmentation","task_name":"Chinese Word Segmentation"},{"task_slug":"deep-learning","task_name":"Deep Learning"},{"task_slug":"sentiment-analysis","task_name":"Sentiment Analysis"}],"methods":[],"datasets_introduced":[{"slug":"lsicc","name":"LSICC","full_name":"Large Scale Informal Chinese Corpus"}],"methods_introduced":[],"results":[],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}