{"url":"/task/chinese-word-segmentation","name":"Chinese Word Segmentation","slug":"chinese-word-segmentation","description_markdown":"Chinese word segmentation is the task of splitting Chinese text (i.e. a sequence of Chinese characters) into words (Source: www.nlpprogress.com).","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":244,"papers_with_code":50,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/chinese-word-segmentation-on-msr","slug":"chinese-word-segmentation-on-msr","dataset":"MSR","dataset_url":null,"rows_in_archive":6,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"BABERT-LE","paper_title":"Unsupervised Boundary-Aware Language Model Pretraining for Chinese Sequence Labeling","paper_url":"/paper/unsupervised-boundary-aware-language-model","paper_date":"2022-10-27","arxiv_id":"2210.15231","code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"modelscope/AdaSeq","url":"https://github.com/modelscope/AdaSeq/tree/master/examples/babert"}],"syntology":null}},{"leaderboard":"/sota/chinese-word-segmentation-on-ctb6","slug":"chinese-word-segmentation-on-ctb6","dataset":"CTB6","dataset_url":null,"rows_in_archive":4,"metrics":["F1"],"first_row_in_archive_order":{"model":"LATTE (Linguistic units, lattices, PTMs, GNNs)","paper_title":"LATTE: Lattice ATTentive Encoding for Character-based Word Segmentation","paper_url":"/paper/latte-lattice-attentive-encoding-for","paper_date":"2023-06-01","arxiv_id":null,"code_links":[{"title":"tchayintr/latte-ptm-ws","url":"https://github.com/tchayintr/latte-ptm-ws"},{"title":"tchayintr/latte-ws","url":"https://github.com/tchayintr/latte-ws"}],"syntology":null}},{"leaderboard":"/sota/chinese-word-segmentation-on-pku","slug":"chinese-word-segmentation-on-pku","dataset":"PKU","dataset_url":null,"rows_in_archive":4,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"BABERT-LE","paper_title":"Unsupervised Boundary-Aware Language Model Pretraining for Chinese Sequence Labeling","paper_url":"/paper/unsupervised-boundary-aware-language-model","paper_date":"2022-10-27","arxiv_id":"2210.15231","code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"modelscope/AdaSeq","url":"https://github.com/modelscope/AdaSeq/tree/master/examples/babert"}],"syntology":null}},{"leaderboard":"/sota/chinese-word-segmentation-on-msra","slug":"chinese-word-segmentation-on-msra","dataset":"MSRA","dataset_url":"/dataset/msra-cn-ner","rows_in_archive":3,"metrics":["F1"],"first_row_in_archive_order":{"model":"BABERT-LE","paper_title":"Unsupervised Boundary-Aware Language Model Pretraining for Chinese Sequence Labeling","paper_url":"/paper/unsupervised-boundary-aware-language-model","paper_date":"2022-10-27","arxiv_id":"2210.15231","code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"modelscope/AdaSeq","url":"https://github.com/modelscope/AdaSeq/tree/master/examples/babert"}],"syntology":null}},{"leaderboard":"/sota/chinese-word-segmentation-on-as","slug":"chinese-word-segmentation-on-as","dataset":"AS","dataset_url":null,"rows_in_archive":2,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"Glyce + BERT","paper_title":"Glyce: Glyph-vectors for Chinese Character Representations","paper_url":"/paper/glyce-glyph-vectors-for-chinese-character","paper_date":"2019-01-29","arxiv_id":"1901.10125","code_links":[{"title":"ShannonAI/glyce","url":"https://github.com/ShannonAI/glyce"},{"title":"zhangyuwangumass/Glyph-based-Chinese-Character-Embedding","url":"https://github.com/zhangyuwangumass/Glyph-based-Chinese-Character-Embedding"}],"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}}},{"leaderboard":"/sota/chinese-word-segmentation-on-cityu","slug":"chinese-word-segmentation-on-cityu","dataset":"CITYU","dataset_url":null,"rows_in_archive":2,"metrics":["F1","Precision","Recall"],"first_row_in_archive_order":{"model":"WMSeg + ZEN","paper_title":"Improving Chinese Word Segmentation with Wordhood Memory Networks","paper_url":"/paper/improving-chinese-word-segmentation-with","paper_date":"2020-07-01","arxiv_id":null,"code_links":[{"title":"SVAIGBA/WMSeg","url":"https://github.com/SVAIGBA/WMSeg"}],"syntology":null}}],"datasets":[{"url":"/dataset/msra-cn-ner","name":"MSRA CN NER","full_name":"MSRA CN NER Dataset","num_papers_in_archive":23},{"url":"/dataset/cuge","name":"CUGE","full_name":"","num_papers_in_archive":4},{"url":"/dataset/lsicc","name":"LSICC","full_name":"Large Scale Informal Chinese Corpus","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/chinese","name":"Chinese"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":50,"tagged_in_all":244,"items":[{"url":"/paper/zen-pre-training-chinese-text-encoder","title":"ZEN: Pre-training Chinese Text Encoder Enhanced by N-gram Representations","date":"2019-11-02","arxiv_id":"1911.00720","repositories_listed":7,"syntology":{"n":35,"n_ran":7,"n_unverified":28,"n_pointer_only":0}},{"url":"/paper/pkuseg-a-toolkit-for-multi-domain-chinese","title":"PKUSEG: A Toolkit for Multi-Domain Chinese Word Segmentation","date":"2019-06-27","arxiv_id":"1906.11455","repositories_listed":4,"syntology":null},{"url":"/paper/simplifying-neural-machine-translation-with","title":"Simplifying Neural Machine Translation with Addition-Subtraction Twin-Gated Recurrent Networks","date":"2018-10-30","arxiv_id":"1810.12546","repositories_listed":3,"syntology":null},{"url":"/paper/latte-lattice-attentive-encoding-for","title":"LATTE: Lattice ATTentive Encoding for Character-based Word Segmentation","date":"2023-06-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-boundary-aware-language-model","title":"Unsupervised Boundary-Aware Language Model Pretraining for Chinese Sequence Labeling","date":"2022-10-27","arxiv_id":"2210.15231","repositories_listed":2,"syntology":null},{"url":"/paper/shuowen-jiezi-linguistically-informed","title":"Sub-Character Tokenization for Chinese Pretrained Language Models","date":"2021-06-01","arxiv_id":"2106.00400","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/investigating-self-attention-network-for","title":"Investigating Self-Attention Network for Chinese Word Segmentation","date":"2019-07-26","arxiv_id":"1907.11512","repositories_listed":2,"syntology":null},{"url":"/paper/glyce-glyph-vectors-for-chinese-character","title":"Glyce: Glyph-vectors for Chinese Character Representations","date":"2019-01-29","arxiv_id":"1901.10125","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/lsicc-a-large-scale-informal-chinese-corpus","title":"LSICC: A Large Scale Informal Chinese Corpus","date":"2018-11-26","arxiv_id":"1811.10167","repositories_listed":2,"syntology":null},{"url":"/paper/segmental-recurrent-neural-networks","title":"Segmental Recurrent Neural Networks","date":"2015-11-18","arxiv_id":"1511.06018","repositories_listed":2,"syntology":null},{"url":"/paper/mining-word-boundaries-from-speech-text","title":"Mining Word Boundaries from Speech-Text Parallel Data for Cross-domain Chinese Word Segmentation","date":"2024-12-12","arxiv_id":"2412.09045","repositories_listed":1,"syntology":null},{"url":"/paper/legal-documents-drafting-with-fine-tuned-pre","title":"Legal Documents Drafting with Fine-Tuned Pre-Trained Large Language Model","date":"2024-06-06","arxiv_id":"2406.04202","repositories_listed":1,"syntology":null},{"url":"/paper/the-uncertainty-based-retrieval-framework-for-1","title":"The Uncertainty-based Retrieval Framework for Ancient Chinese CWS and POS","date":"2023-10-12","arxiv_id":"2310.08496","repositories_listed":1,"syntology":null},{"url":"/paper/ancient-chinese-word-segmentation-and-part-of-1","title":"Ancient Chinese Word Segmentation and Part-of-Speech Tagging Using Distant Supervision","date":"2023-03-03","arxiv_id":"2303.01912","repositories_listed":1,"syntology":null},{"url":"/paper/gnn-sl-sequence-labeling-based-on-nearest","title":"GNN-SL: Sequence Labeling Based on Nearest Examples via GNN","date":"2022-12-05","arxiv_id":"2212.02017","repositories_listed":1,"syntology":null},{"url":"/paper/chimst-a-chinese-medical-corpus-for-word","title":"ChiMST: A Chinese Medical Corpus for Word Segmentation and Medical Term Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/weighted-self-distillation-for-chinese-word","title":"Weighted self Distillation for Chinese word segmentation","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/word-segmentation-by-separation-inference-for","title":"Word Segmentation by Separation Inference for East Asian Languages","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-chinese-word-segmentation-with-1","title":"Unsupervised Chinese Word Segmentation with BERT Oriented Probing and Transformation","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fast-and-accurate-span-based-semantic-role","title":"Fast and Accurate End-to-End Span-based Semantic Role Labeling as Word-based Graph Parsing","date":"2021-12-06","arxiv_id":"2112.02970","repositories_listed":1,"syntology":null},{"url":"/paper/more-than-text-multi-modal-chinese-word","title":"More than Text: Multi-modal Chinese Word Segmentation","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploring-word-segmentation-and-medical","title":"Exploring Word Segmentation and Medical Concept Recognition for Chinese Medical Texts","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/joint-chinese-word-segmentation-and-part-of-1","title":"Joint Chinese Word Segmentation and Part-of-speech Tagging via Multi-channel Attention of Character N-grams","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-url-segmentation-with-1","title":"Semi-supervised URL Segmentation with Recurrent Neural Networks Pre-trained on Knowledge Graph Entities","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rethinkcws-is-chinese-word-segmentation-a","title":"RethinkCWS: Is Chinese Word Segmentation a Solved Task?","date":"2020-11-13","arxiv_id":"2011.06858","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-url-segmentation-with","title":"Semi-supervised URL Segmentation with Recurrent Neural NetworksPre-trained on Knowledge Graph Entities","date":"2020-11-05","arxiv_id":"2011.03138","repositories_listed":1,"syntology":null},{"url":"/paper/n-ltp-a-open-source-neural-chinese-language","title":"N-LTP: An Open-source Neural Language Technology Platform for Chinese","date":"2020-09-24","arxiv_id":"2009.11616","repositories_listed":1,"syntology":null},{"url":"/paper/fasthan-a-bert-based-joint-many-task-toolkit","title":"fastHan: A BERT-based Multi-Task Toolkit for Chinese NLP","date":"2020-09-18","arxiv_id":"2009.08633","repositories_listed":1,"syntology":null},{"url":"/paper/approaching-neural-chinese-word-segmentation","title":"Approaching Neural Chinese Word Segmentation as a Low-Resource Machine Translation Task","date":"2020-08-12","arxiv_id":"2008.05348","repositories_listed":1,"syntology":null},{"url":"/paper/coupling-distant-annotation-and-adversarial-1","title":"Coupling Distant Annotation and Adversarial Training for Cross-Domain Chinese Word Segmentation","date":"2020-07-16","arxiv_id":"2007.08186","repositories_listed":1,"syntology":null}],"syntology_records":3,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}