{"url":"/method/awd-lstm","slug":"awd-lstm","name":"AWD-LSTM","full_name":"ASGD Weight-Dropped LSTM","full_name_withheld":false,"description_markdown":"**ASGD Weight-Dropped LSTM**, or **AWD-LSTM**, is a type of recurrent neural network that employs [DropConnect](https://paperswithcode.com/method/dropconnect) for regularization, as well as [NT-ASGD](https://paperswithcode.com/method/nt-asgd) for optimization - non-monotonically triggered averaged [SGD](https://paperswithcode.com/method/sgd) - which returns an average of last iterations of weights. Additional regularization techniques employed include variable length backpropagation sequences, [variational dropout](https://paperswithcode.com/method/variational-dropout), [embedding dropout](https://paperswithcode.com/method/embedding-dropout), [weight tying](https://paperswithcode.com/method/weight-tying), independent embedding/hidden size, [activation regularization](https://paperswithcode.com/method/activation-regularization) and [temporal activation regularization](https://paperswithcode.com/method/temporal-activation-regularization).","description_state":"present","introduced_year":null,"introduced_by":{"title":"Regularizing and Optimizing LSTM Language Models","paper":"/paper/regularizing-and-optimizing-lstm-language","first_author":"Stephen Merity","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/regularizing-and-optimizing-lstm-language"},"source":{"url":"http://arxiv.org/abs/1708.02182v1","title":"Regularizing and Optimizing LSTM Language Models","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Sequential","area_id":"sequential","collection":"Recurrent Neural Networks","url":"/methods/category/recurrent-neural-networks","pwc_aliases":[]}],"n_papers_tagged":52,"archive_num_papers":52,"papers_newest_first":[{"paper":null,"title":"Advanced Deep Learning Techniques for Analyzing Earnings Call Transcripts: Methodologies and Applications","date":"2025-02-27","arxiv_id":"2503.01886","n_code_links":0,"syntology":null},{"paper":null,"title":"No Argument Left Behind: Overlapping Chunks for Faster Processing of Arbitrarily Long Legal Texts","date":"2024-10-24","arxiv_id":"2410.19184","n_code_links":0,"syntology":null},{"paper":"/paper/rico-reddit-ideological-communities","title":"RICo: Reddit ideological communities","date":"2024-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/exploring-multi-level-threats-in-telegram","title":"Exploring Multi-Level Threats in Telegram Data with AI-Human Annotation: A Preliminary Study","date":"2023-12-15","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Illicit Darkweb Classification via Natural-language Processing: Classifying Illicit Content of Webpages based on Textual Information","date":"2023-12-08","arxiv_id":"2312.04944","n_code_links":0,"syntology":null},{"paper":null,"title":"Explainable and High-Performance Hate and Offensive Speech Detection","date":"2022-06-26","arxiv_id":"2206.12983","n_code_links":0,"syntology":null},{"paper":"/paper/iiitt-dravidian-codemix-fire2021","title":"IIITT@Dravidian-CodeMix-FIRE2021: Transliterate or translate? Sentiment analysis of code-mixed text in Dravidian languages","date":"2021-11-15","arxiv_id":"2111.07906","n_code_links":1,"syntology":null},{"paper":"/paper/offensive-language-identification-in-low","title":"Offensive Language Identification in Low-resourced Code-mixed Dravidian languages using Pseudo-labeling","date":"2021-08-27","arxiv_id":"2108.12177","n_code_links":1,"syntology":null},{"paper":"/paper/sn-computer-science-towards-offensive","title":"Towards Offensive Language Identification for Tamil Code-Mixed YouTube Comments and Posts","date":"2021-08-24","arxiv_id":"2108.10939","n_code_links":1,"syntology":null},{"paper":null,"title":"Learning ULMFiT and Self-Distillation with Calibration for Medical Dialogue System","date":"2021-07-20","arxiv_id":"2107.09625","n_code_links":0,"syntology":null},{"paper":"/paper/whose-heritage-classification-of-unesco-world","title":"WHOSe Heritage: Classification of UNESCO World Heritage \"Outstanding Universal Value\" Documents with Soft Labels","date":"2021-04-12","arxiv_id":"2104.05547","n_code_links":1,"syntology":null},{"paper":"/paper/l3cubemahasent-a-marathi-tweet-based","title":"L3CubeMahaSent: A Marathi Tweet-based Sentiment Analysis Dataset","date":"2021-03-21","arxiv_id":"2103.11408","n_code_links":1,"syntology":null},{"paper":"/paper/indicnlp-kgp-at-dravidianlangtech-eacl2021","title":"indicnlp@kgp at DravidianLangTech-EACL2021: Offensive Language Identification in Dravidian Languages","date":"2021-02-14","arxiv_id":"2102.07150","n_code_links":1,"syntology":null},{"paper":"/paper/indicnlp-kgp-at-dravidianlangtech-eacl2021-1","title":"indicnlp@ kgp at DravidianLangTech-EACL2021: Offensive Language Identification in Dravidian Languages","date":"2021-02-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Train your classifier first: Cascade Neural Networks Training from upper layers to lower layers","date":"2021-02-09","arxiv_id":"2102.04697","n_code_links":0,"syntology":null},{"paper":null,"title":"Experimental Evaluation of Deep Learning models for Marathi Text Classification","date":"2021-01-13","arxiv_id":"2101.04899","n_code_links":0,"syntology":null},{"paper":"/paper/ladiff-ulmfit-a-layer-differentiated-training","title":"LaDiff ULMFiT: A Layer Differentiated training approach for ULMFiT","date":"2021-01-13","arxiv_id":"2101.04965","n_code_links":1,"syntology":null},{"paper":null,"title":"Post-Training Weighted Quantization of Neural Networks for Language Models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/hinglishnlp-at-semeval-2020-task-9-fine-tuned","title":"HinglishNLP at SemEval-2020 Task 9: Fine-tuned Language Models for Hinglish Sentiment Detection","date":"2020-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"title":"Smash at SemEval-2020 Task 7: Optimizing the Hyperparameters of ERNIE 2.0 for Humor Ranking and Rating","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Palomino-Ochoa at SemEval-2020 Task 9: Robust System based on Transformer for Code-Mixed Sentiment Classification","date":"2020-11-18","arxiv_id":"2011.09448","n_code_links":0,"syntology":null},{"paper":"/paper/pagsusuri-ng-rnn-based-transfer-learning","title":"Pagsusuri ng RNN-based Transfer Learning Technique sa Low-Resource Language","date":"2020-10-13","arxiv_id":"2010.06447","n_code_links":2,"syntology":null},{"paper":null,"title":"Gauravarora@HASOC-Dravidian-CodeMix-FIRE2020: Pre-training ULMFiT on Synthetically Generated Code-Mixed Data for Hate Speech Detection","date":"2020-10-05","arxiv_id":"2010.02094","n_code_links":0,"syntology":null},{"paper":null,"title":"Fine-tuning Pre-trained Contextual Embeddings for Citation Content Analysis in Scholarly Publication","date":"2020-09-12","arxiv_id":"2009.05836","n_code_links":0,"syntology":null},{"paper":"/paper/hinglishnlp-fine-tuned-language-models-for","title":"HinglishNLP: Fine-tuned Language Models for Hinglish Sentiment Detection","date":"2020-08-22","arxiv_id":"2008.09820","n_code_links":2,"syntology":null},{"paper":"/paper/composer-style-classification-of-piano-sheet","title":"Composer Style Classification of Piano Sheet Music Images Using Language Model Pretraining","date":"2020-07-29","arxiv_id":"2007.14587","n_code_links":1,"syntology":null},{"paper":null,"title":"Probing for Referential Information in Language Models","date":"2020-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Evaluation Metrics for Headline Generation Using Deep Pre-Trained Embeddings","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/offensive-language-detection-in-arabic-using","title":"Offensive language detection in Arabic using ULMFiT","date":"2020-05-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Text Categorization for Conflict Event Annotation","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":19},{"task":"/task/language-modeling","name":"Language Modeling","papers":17},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":16},{"task":"/task/text-classification","name":"Text Classification","papers":15},{"task":"/task/classification","name":"General Classification","papers":14},{"task":"/task/text-classification-1","name":"text-classification","papers":11},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":9},{"task":"/task/classification-1","name":"Classification","papers":8},{"task":"/task/language-identification","name":"Language Identification","papers":4},{"task":"/task/translation","name":"Translation","papers":4},{"task":"/task/word-embeddings","name":"Word Embeddings","papers":4},{"task":"/task/decision-making","name":"Decision Making","papers":3},{"task":"/task/hate-speech-detection","name":"Hate Speech Detection","papers":3},{"task":"/task/machine-translation","name":"Machine Translation","papers":3},{"task":"/task/sentence","name":"Sentence","papers":3},{"task":"/task/sentiment-classification","name":"Sentiment Classification","papers":3},{"task":"/task/articles","name":"Articles","papers":2},{"task":"/task/machine-learning","name":"BIG-bench Machine Learning","papers":2},{"task":"/task/image-classification","name":"Image Classification","papers":2},{"task":"/task/management","name":"Management","papers":2}],"tasks_shown":20,"n_tasks":65,"usage_by_year":[{"year":"2017","papers":2},{"year":"2018","papers":3},{"year":"2019","papers":14},{"year":"2020","papers":15},{"year":"2021","papers":12},{"year":"2022","papers":1},{"year":"2023","papers":2},{"year":"2024","papers":2},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/awd-lstm"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}