{"url":"/method/canine","slug":"canine","name":"CANINE","full_name":"CANINE","full_name_withheld":false,"description_markdown":"**CANINE** is a pre-trained encoder for language understanding that operates directly on character sequences—without explicit tokenization or vocabulary—and a pre-training strategy with soft inductive biases in place of hard token boundaries. To use its finer-grained input effectively and efficiently, Canine combines downsampling, which reduces the input sequence length, with a deep [transformer](https://paperswithcode.com/method/transformer) stack, which encodes context.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/2103.06874v4","title":"CANINE: Pre-training an Efficient Tokenization-Free Encoder for Language Representation","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":6,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"Token-free Models for Sarcasm Detection","date":"2025-05-02","arxiv_id":"2505.01006","n_code_links":0,"syntology":null},{"paper":null,"title":"Phishing Website Detection through Multi-Model Analysis of HTML Content","date":"2024-01-09","arxiv_id":"2401.04820","n_code_links":0,"syntology":null},{"paper":null,"title":"SecureReg: Combining NLP and MLP for Enhanced Detection of Malicious Domain Name Registrations","date":"2024-01-06","arxiv_id":"2401.03196","n_code_links":0,"syntology":null},{"paper":null,"title":"Elementwise Language Representation","date":"2023-02-27","arxiv_id":"2302.13475","n_code_links":0,"syntology":null},{"paper":"/paper/an-ensemble-of-pre-trained-transformer-models","title":"An Ensemble of Pre-trained Transformer Models For Imbalanced Multiclass Malware Classification","date":"2021-12-25","arxiv_id":"2112.13236","n_code_links":1,"syntology":null},{"paper":"/paper/canine-pre-training-an-efficient-tokenization","title":"CANINE: Pre-training an Efficient Tokenization-Free Encoder for Language Representation","date":"2021-03-11","arxiv_id":"2103.06874","n_code_links":6,"syntology":null}],"papers_shown":6,"tasks":[{"task":"/task/document-classification","name":"Document Classification","papers":1},{"task":"/task/inductive-bias","name":"Inductive Bias","papers":1},{"task":"/task/malware-classification","name":"Malware Classification","papers":1},{"task":"/task/phishing-website-detection","name":"Phishing Website Detection","papers":1},{"task":"/task/sarcasm-detection","name":"Sarcasm Detection","papers":1},{"task":"/task/specificity","name":"Specificity","papers":1}],"tasks_shown":6,"n_tasks":6,"usage_by_year":[{"year":"2021","papers":2},{"year":"2023","papers":1},{"year":"2024","papers":2},{"year":"2025","papers":1}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/canine"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}