{"url":"/task/spelling-correction","name":"Spelling Correction","slug":"spelling-correction","description_markdown":"Spelling correction is the task of detecting and correcting spelling mistakes.","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":193,"papers_with_code":52,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":1,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/github-typo-corpus","name":"GitHub Typo Corpus","full_name":"GitHub Typo Corpus","num_papers_in_archive":3},{"url":"/dataset/cscd-ime","name":"CSCD-IME","full_name":"","num_papers_in_archive":2},{"url":"/dataset/mcscset","name":"MCSCSet","full_name":"","num_papers_in_archive":2},{"url":"/dataset/viwiki-spelling","name":"Viwiki-Spelling","full_name":"Vietnamese Spelling Correction Dataset","num_papers_in_archive":1}],"subtasks":[{"url":"/task/bangla-spelling-error-correction","name":"Bangla Spelling Error Correction"}],"parent_tasks":[{"url":"/task/text-generation","name":"Text Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":52,"tagged_in_all":193,"items":[{"url":"/paper/an-actor-critic-algorithm-for-sequence","title":"An Actor-Critic Algorithm for Sequence Prediction","date":"2016-07-24","arxiv_id":"1607.07086","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/a-methodology-for-generative-spelling","title":"A Methodology for Generative Spelling Correction via Natural Spelling Errors Emulation across Multiple Domains and Languages","date":"2023-08-18","arxiv_id":"2308.09435","repositories_listed":2,"syntology":null},{"url":"/paper/chinese-spelling-correction-as-rephrasing","title":"Chinese Spelling Correction as Rephrasing Language Model","date":"2023-08-17","arxiv_id":"2308.08796","repositories_listed":2,"syntology":null},{"url":"/paper/an-extended-sequence-tagging-vocabulary-for","title":"An Extended Sequence Tagging Vocabulary for Grammatical Error Correction","date":"2023-02-12","arxiv_id":"2302.05913","repositories_listed":2,"syntology":null},{"url":"/paper/tokenization-repair-in-the-presence-of","title":"Tokenization Repair in the Presence of Spelling Errors","date":"2020-10-15","arxiv_id":"2010.07878","repositories_listed":2,"syntology":null},{"url":"/paper/robust-to-noise-models-in-natural-language","title":"Robust to Noise Models in Natural Language Processing Tasks","date":"2019-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/monoise-modeling-noise-using-a-modular","title":"MoNoise: Modeling Noise Using a Modular Normalization System","date":"2017-10-10","arxiv_id":"1710.03476","repositories_listed":2,"syntology":null},{"url":"/paper/tispell-a-semi-masked-methodology-for-tibetan","title":"TiSpell: A Semi-Masked Methodology for Tibetan Spelling Correction covering Multi-Level Error with Data Augmentation","date":"2025-05-12","arxiv_id":"2505.08037","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-impact-of-multimodal-features","title":"Unveiling the Impact of Multimodal Features on Chinese Spelling Correction: From Analysis to Design","date":"2025-04-10","arxiv_id":"2504.07661","repositories_listed":1,"syntology":null},{"url":"/paper/a-training-free-llm-based-approach-to-general","title":"A Training-free LLM-based Approach to General Chinese Character Error Correction","date":"2025-02-21","arxiv_id":"2502.15266","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-character-level-understanding-in","title":"Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning","date":"2024-11-26","arxiv_id":"2411.17679","repositories_listed":1,"syntology":null},{"url":"/paper/cnmbert-a-model-for-hanyu-pinyin-abbreviation","title":"CNMBERT: A Model for Converting Hanyu Pinyin Abbreviations to Chinese Characters","date":"2024-11-18","arxiv_id":"2411.11770","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-yet-effective-training-free-prompt","title":"A Simple yet Effective Training-free Prompt-free Approach to Chinese Spelling Correction Based on Large Language Models","date":"2024-10-05","arxiv_id":"2410.04027","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/edacsc-two-easy-data-augmentation-methods-for","title":"EdaCSC: Two Easy Data Augmentation Methods for Chinese Spelling Correction","date":"2024-09-08","arxiv_id":"2409.05105","repositories_listed":1,"syntology":null},{"url":"/paper/a-coin-has-two-sides-a-novel-detector","title":"A Coin Has Two Sides: A Novel Detector-Corrector Framework for Chinese Spelling Correction","date":"2024-09-06","arxiv_id":"2409.04150","repositories_listed":1,"syntology":null},{"url":"/paper/araspell-a-deep-learning-approach-for-arabic","title":"AraSpell: A Deep Learning Approach for Arabic Spelling Correction","date":"2024-05-11","arxiv_id":"2405.06981","repositories_listed":1,"syntology":null},{"url":"/paper/eval-gcsc-a-new-metric-for-evaluating-chatgpt","title":"Eval-GCSC: A New Metric for Evaluating ChatGPT's Performance in Chinese Spelling Correction","date":"2023-11-14","arxiv_id":"2311.08219","repositories_listed":1,"syntology":null},{"url":"/paper/gio-gradient-information-optimization-for","title":"GIO: Gradient Information Optimization for Training Dataset Selection","date":"2023-06-20","arxiv_id":"2306.11670","repositories_listed":1,"syntology":null},{"url":"/paper/spellmapper-a-non-autoregressive-neural","title":"SpellMapper: A non-autoregressive neural spellchecker for ASR customization with candidate retrieval based on n-gram mappings","date":"2023-06-04","arxiv_id":"2306.02317","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-masked-language-modeling-for","title":"Rethinking Masked Language Modeling for Chinese Spelling Correction","date":"2023-05-28","arxiv_id":"2305.17721","repositories_listed":1,"syntology":null},{"url":"/paper/disentangled-phonetic-representation-for","title":"Disentangled Phonetic Representation for Chinese Spelling Correction","date":"2023-05-24","arxiv_id":"2305.14783","repositories_listed":1,"syntology":null},{"url":"/paper/an-error-guided-correction-model-for-chinese","title":"An Error-Guided Correction Model for Chinese Spelling Error Correction","date":"2023-01-16","arxiv_id":"2301.06323","repositories_listed":1,"syntology":null},{"url":"/paper/inducing-character-level-structure-in-subword","title":"Inducing Character-level Structure in Subword-based Language Models with Type-level Interchange Intervention Training","date":"2022-12-19","arxiv_id":"2212.09897","repositories_listed":1,"syntology":null},{"url":"/paper/cscd-ime-correcting-spelling-errors-generated","title":"CSCD-NS: a Chinese Spelling Check Dataset for Native Speakers","date":"2022-11-16","arxiv_id":"2211.08788","repositories_listed":1,"syntology":null},{"url":"/paper/mcscset-a-specialist-annotated-dataset-for","title":"MCSCSet: A Specialist-annotated Dataset for Medical-domain Chinese Spelling Correction","date":"2022-10-21","arxiv_id":"2210.11720","repositories_listed":1,"syntology":null},{"url":"/paper/look-ma-only-400-samples-revisiting-the","title":"Look Ma, Only 400 Samples! Revisiting the Effectiveness of Automatic N-Gram Rule Generation for Spelling Normalization in Filipino","date":"2022-10-06","arxiv_id":"2210.02675","repositories_listed":1,"syntology":null},{"url":"/paper/bspell-a-cnn-blended-bert-based-bengali-spell","title":"BSpell: A CNN-Blended BERT Based Bangla Spell Checker","date":"2022-08-20","arxiv_id":"2208.09709","repositories_listed":1,"syntology":null},{"url":"/paper/abb-bert-a-bert-model-for-disambiguating","title":"ABB-BERT: A BERT model for disambiguating abbreviations and contractions","date":"2022-07-08","arxiv_id":"2207.04008","repositories_listed":1,"syntology":null},{"url":"/paper/craspell-a-contextual-typo-robust-approach-to","title":"CRASpell: A Contextual Typo Robust Approach to Improve Chinese Spelling Correction","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-contextual-spelling-correction-for","title":"Towards Contextual Spelling Correction for Customization of End-to-end Speech Recognition Systems","date":"2022-03-02","arxiv_id":"2203.00888","repositories_listed":1,"syntology":null}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}