{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/a-fast-and-accurate-vietnamese-word-segmenter","title":"A Fast and Accurate Vietnamese Word Segmenter","arxiv_id":"1709.06307","date":"2017-09-19","proceeding":"LREC 2018 5","authors":["Dat Quoc Nguyen","Dai Quoc Nguyen","Thanh Vu","Mark Dras","Mark Johnson"],"abstract":"We propose a novel approach to Vietnamese word segmentation. Our approach is\nbased on the Single Classification Ripple Down Rules methodology (Compton and\nJansen, 1990), where rules are stored in an exception structure and new rules\nare only added to correct segmentation errors given by existing rules.\nExperimental results on the benchmark Vietnamese treebank show that our\napproach outperforms previous state-of-the-art approaches JVnSegmenter,\nvnTokenizer, DongDu and UETsegmenter in terms of both accuracy and performance\nspeed. Our code is open-source and available at:\nhttps://github.com/datquocnguyen/RDRsegmenter.","url_abs":"http://arxiv.org/abs/1709.06307v2","url_pdf":"http://arxiv.org/pdf/1709.06307v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"a-fast-and-accurate-vietnamese-word-segmenter","repo_url":"https://github.com/datquocnguyen/RDRsegmenter","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"classification","task_name":"General Classification"},{"task_slug":"segmentation","task_name":"Segmentation"},{"task_slug":"vietnamese-word-segmentation","task_name":"Vietnamese Word Segmentation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1709.06307","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}