{"url":"/method/charformer","slug":"charformer","name":"Charformer","full_name":"Charformer","full_name_withheld":false,"description_markdown":"**Charformer** is a type of [Transformer](https://paperswithcode.com/methods/category/transformers) model that learns a subword tokenization end-to-end as part of the model. Specifically it uses [GBST](https://paperswithcode.com/method/gradient-based-subword-tokenization) that automatically learns latent subword representations from characters in a data-driven fashion. Following GBST, the soft subword sequence is passed through [Transformer](https://paperswithcode.com/method/transformer) layers.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Charformer: Fast Character Transformers via Gradient-based Subword Tokenization","paper":"/paper/charformer-fast-character-transformers-via","first_author":"Yi Tay","n_authors":10,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/charformer-fast-character-transformers-via"},"source":{"url":"https://arxiv.org/abs/2106.12672v3","title":"Charformer: Fast Character Transformers via Gradient-based Subword Tokenization","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":5,"archive_num_papers":5,"papers_newest_first":[{"paper":"/paper/charformer-a-glyph-fusion-based-attentive","title":"CharFormer: A Glyph Fusion based Attentive Framework for High-precision Character Image Denoising","date":"2022-07-16","arxiv_id":"2207.07798","n_code_links":1,"syntology":null},{"paper":"/paper/patching-leaks-in-the-charformer-for-1","title":"Patching Leaks in the Charformer for Efficient Character-Level Generation","date":"2022-05-27","arxiv_id":"2205.14086","n_code_links":1,"syntology":null},{"paper":null,"title":"A New Generation of Perspective API: Efficient Multilingual Character-level Transformers","date":"2022-02-22","arxiv_id":"2202.11176","n_code_links":0,"syntology":null},{"paper":null,"title":"Patching Leaks in the Charformer for Generative Tasks","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/charformer-fast-character-transformers-via","title":"Charformer: Fast Character Transformers via Gradient-based Subword Tokenization","date":"2021-06-23","arxiv_id":"2106.12672","n_code_links":2,"syntology":{"ran":7,"of":10,"unverified":3,"pointer_only":0}}],"papers_shown":5,"tasks":[{"task":"/task/decoder","name":"Decoder","papers":2},{"task":"/task/nmt","name":"NMT","papers":2},{"task":"/task/denoising","name":"Denoising","papers":1},{"task":"/task/image-denoising","name":"Image Denoising","papers":1},{"task":"/task/inductive-bias","name":"Inductive Bias","papers":1},{"task":"/task/linguistic-acceptability","name":"Linguistic Acceptability","papers":1},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":1},{"task":"/task/paraphrase-identification","name":"Paraphrase Identification","papers":1},{"task":"/task/semantic-textual-similarity","name":"Semantic Textual Similarity","papers":1},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":1},{"task":"/task/toxic-comment-classification","name":"Toxic Comment Classification","papers":1},{"task":"/task/translation","name":"Translation","papers":1}],"tasks_shown":12,"n_tasks":12,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":4}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/charformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}