{"url":"/method/fasttext","slug":"fasttext","name":"fastText","full_name":"fastText","full_name_withheld":false,"description_markdown":"**fastText** embeddings exploit subword information to construct word embeddings. Representations are learnt of character $n$-grams, and words represented as the sum of the $n$-gram vectors. This extends the word2vec type models with subword information. This helps the embeddings understand suffixes and prefixes. Once a word is represented using character $n$-grams, a skipgram model is trained to learn the embeddings.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Enriching Word Vectors with Subword Information","paper":"/paper/enriching-word-vectors-with-subword","first_author":"Piotr Bojanowski","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/enriching-word-vectors-with-subword"},"source":{"url":"http://arxiv.org/abs/1607.04606v2","title":"Enriching Word Vectors with Subword Information","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Static Word Embeddings","url":"/methods/category/static-word-embeddings","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Word Embeddings","url":"/methods/category/word-embeddings","pwc_aliases":[]}],"n_papers_tagged":240,"archive_num_papers":240,"papers_newest_first":[{"paper":null,"title":"Speak2Sign3D: A Multi-modal Pipeline for English Speech to American Sign Language Animation","date":"2025-07-09","arxiv_id":"2507.06530","n_code_links":0,"syntology":null},{"paper":"/paper/guarded-query-routing-for-large-language","title":"Guarded Query Routing for Large Language Models","date":"2025-05-20","arxiv_id":"2505.14524","n_code_links":1,"syntology":null},{"paper":null,"title":"Ultra-FineWeb: Efficient Data Filtering and Verification for High-Quality LLM Training Data","date":"2025-05-08","arxiv_id":"2505.05427","n_code_links":0,"syntology":null},{"paper":"/paper/myner-contextualized-burmese-named-entity","title":"myNER: Contextualized Burmese Named Entity Recognition with Bidirectional LSTM and fastText Embeddings via Joint Training with POS Tagging","date":"2025-04-05","arxiv_id":"2504.04038","n_code_links":1,"syntology":null},{"paper":"/paper/a-data-driven-investigation-of-euphemistic","title":"A Data-driven Investigation of Euphemistic Language: Comparing the usage of \"slave\" and \"servant\" in 19th century US newspapers","date":"2025-03-19","arxiv_id":"2503.15057","n_code_links":1,"syntology":null},{"paper":null,"title":"Text classification using machine learning methods","date":"2025-02-27","arxiv_id":"2502.19801","n_code_links":0,"syntology":null},{"paper":"/paper/poster-long-php-webshell-files-detection","title":"Poster: Long PHP webshell files detection based on sliding window attention","date":"2025-02-26","arxiv_id":"2502.19257","n_code_links":1,"syntology":null},{"paper":null,"title":"A Multi-tiered Solution for Personalized Baggage Item Recommendations using FastText and Association Rule Mining","date":"2025-01-16","arxiv_id":"2501.09359","n_code_links":0,"syntology":null},{"paper":null,"title":"Sentiment Analysis in Twitter Social Network Centered on Cryptocurrencies Using Machine Learning","date":"2025-01-16","arxiv_id":"2501.09777","n_code_links":0,"syntology":null},{"paper":null,"title":"Research on Violent Text Detection System Based on BERT-fasttext Model","date":"2024-12-21","arxiv_id":"2412.16455","n_code_links":0,"syntology":null},{"paper":null,"title":"UnMA-CapSumT: Unified and Multi-Head Attention-driven Caption Summarization Transformer","date":"2024-12-16","arxiv_id":"2412.11836","n_code_links":0,"syntology":null},{"paper":null,"title":"On Importance of Code-Mixed Embeddings for Hate Speech Identification","date":"2024-11-27","arxiv_id":"2411.18577","n_code_links":0,"syntology":null},{"paper":"/paper/bert-or-fasttext-a-comparative-analysis-of","title":"BERT or FastText? A Comparative Analysis of Contextual as well as Non-Contextual Embeddings","date":"2024-11-26","arxiv_id":"2411.17661","n_code_links":1,"syntology":null},{"paper":null,"title":"From Word Vectors to Multimodal Embeddings: Techniques, Applications, and Future Directions For Large Language Models","date":"2024-11-06","arxiv_id":"2411.05036","n_code_links":0,"syntology":null},{"paper":null,"title":"Generic Embedding-Based Lexicons for Transparent and Reproducible Text Scoring","date":"2024-11-01","arxiv_id":"2411.00964","n_code_links":0,"syntology":null},{"paper":null,"title":"LightFusionRec: Lightweight Transformers-Based Cross-Domain Recommendation Model","date":"2024-10-21","arxiv_id":"2410.15656","n_code_links":0,"syntology":null},{"paper":null,"title":"Stress Detection on Code-Mixed Texts in Dravidian Languages using Machine Learning","date":"2024-10-08","arxiv_id":"2410.06428","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-film-subtitles-is-youtube-the-best","title":"Beyond Film Subtitles: Is YouTube the Best Approximation of Spoken Vocabulary?","date":"2024-10-04","arxiv_id":"2410.03240","n_code_links":1,"syntology":null},{"paper":null,"title":"Individuation in Neural Models with and without Visual Grounding","date":"2024-09-27","arxiv_id":"2409.18868","n_code_links":0,"syntology":null},{"paper":null,"title":"An Evaluation of Sindhi Word Embedding in Semantic Analogies and Downstream Tasks","date":"2024-08-28","arxiv_id":"2408.15720","n_code_links":0,"syntology":null},{"paper":"/paper/2408-01961","title":"Representation Bias of Adolescents in AI: A Bilingual, Bicultural Study","date":"2024-08-04","arxiv_id":"2408.01961","n_code_links":1,"syntology":null},{"paper":null,"title":"Constructing the CORD-19 Vaccine Dataset","date":"2024-07-26","arxiv_id":"2407.18471","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Depressive Post Detection in Bangla: A Comparative Study of TF-IDF, BERT and FastText Embeddings","date":"2024-07-12","arxiv_id":"2407.09187","n_code_links":0,"syntology":null},{"paper":"/paper/masklid-code-switching-language","title":"MaskLID: Code-Switching Language Identification through Iterative Masking","date":"2024-06-10","arxiv_id":"2406.06263","n_code_links":1,"syntology":null},{"paper":"/paper/rico-reddit-ideological-communities","title":"RICo: Reddit ideological communities","date":"2024-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/e2vec-feature-embedding-with-temporal","title":"E2Vec: Feature Embedding with Temporal Information for Analyzing Student Actions in E-Book Systems","date":"2024-05-24","arxiv_id":"2407.13053","n_code_links":1,"syntology":null},{"paper":null,"title":"RE-GrievanceAssist: Enhancing Customer Experience through ML-Powered Complaint Management","date":"2024-04-29","arxiv_id":"2404.18963","n_code_links":0,"syntology":null},{"paper":null,"title":"Deep CNN with late fusion for realtime multimodal emotion recognition","date":"2024-04-15","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/fastspell-the-langid-magic-spell","title":"FastSpell: the LangId Magic Spell","date":"2024-04-12","arxiv_id":"2404.08345","n_code_links":1,"syntology":null},{"paper":"/paper/breaking-the-silence-detecting-and-mitigating","title":"Breaking the Silence Detecting and Mitigating Gendered Abuse in Hindi, Tamil, and Indian English Online Spaces","date":"2024-04-02","arxiv_id":"2404.02013","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/word-embeddings","name":"Word Embeddings","papers":89},{"task":"/task/text-classification","name":"Text Classification","papers":38},{"task":"/task/classification","name":"General Classification","papers":28},{"task":"/task/sentence","name":"Sentence","papers":28},{"task":"/task/text-classification-1","name":"text-classification","papers":27},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":24},{"task":"/task/classification-1","name":"Classification","papers":17},{"task":"/task/named-entity-recognition-1","name":"Named Entity Recognition","papers":17},{"task":"/task/named-entity-recognition","name":"named-entity-recognition","papers":17},{"task":"/task/named-entity-recognition-ner","name":"Named Entity Recognition (NER)","papers":16},{"task":"/task/language-modeling","name":"Language Modeling","papers":12},{"task":"/task/language-modelling","name":"Language Modelling","papers":12},{"task":"/task/word-similarity","name":"Word Similarity","papers":11},{"task":"/task/deep-learning","name":"Deep Learning","papers":9},{"task":"/task/pos","name":"POS","papers":9},{"task":"/task/pos-tagging","name":"POS Tagging","papers":8},{"task":"/task/question-answering","name":"Question Answering","papers":8},{"task":"/task/machine-learning","name":"BIG-bench Machine Learning","papers":7},{"task":"/task/clustering","name":"Clustering","papers":7},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":7}],"tasks_shown":20,"n_tasks":214,"usage_by_year":[{"year":"2016","papers":2},{"year":"2017","papers":12},{"year":"2018","papers":13},{"year":"2019","papers":41},{"year":"2020","papers":54},{"year":"2021","papers":45},{"year":"2022","papers":20},{"year":"2023","papers":19},{"year":"2024","papers":25},{"year":"2025","papers":9}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/fasttext"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}