{"url":"/method/variational-dropout","slug":"variational-dropout","name":"Variational Dropout","full_name":"Variational Dropout","full_name_withheld":false,"description_markdown":"**Variational Dropout** is a regularization technique based on [dropout](https://paperswithcode.com/method/dropout), but uses a variational inference grounded approach. In Variational Dropout, we repeat the same dropout mask at each time step for both inputs, outputs, and recurrent layers (drop the same network units at each time step). This is in contrast to ordinary Dropout where different dropout masks are sampled at each time step for the inputs and outputs alone.","description_state":"present","introduced_year":null,"introduced_by":{"title":"A Theoretically Grounded Application of Dropout in Recurrent Neural Networks","paper":"/paper/a-theoretically-grounded-application-of","first_author":"Yarin Gal","n_authors":2,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/a-theoretically-grounded-application-of"},"source":{"url":"http://arxiv.org/abs/1512.05287v5","title":"A Theoretically Grounded Application of Dropout in Recurrent Neural Networks","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/salesforce/awd-lstm-lm/blob/32fcb42562aeb5c7e6c9dec3f2a3baaaf68a5cb5/weight_drop.py#L5","code_snippet_url_on_a_code_host":true,"categories":[{"area":"General","area_id":"general","collection":"Regularization","url":"/methods/category/regularization","pwc_aliases":[]}],"n_papers_tagged":141,"archive_num_papers":141,"papers_newest_first":[{"paper":"/paper/rlbenchnet-the-right-network-for-the-right","title":"RLBenchNet: The Right Network for the Right Reinforcement Learning Task","date":"2025-05-21","arxiv_id":"2505.15040","n_code_links":1,"syntology":null},{"paper":null,"title":"Advanced Deep Learning Techniques for Analyzing Earnings Call Transcripts: Methodologies and Applications","date":"2025-02-27","arxiv_id":"2503.01886","n_code_links":0,"syntology":null},{"paper":null,"title":"BARNN: A Bayesian Autoregressive and Recurrent Neural Network","date":"2025-01-30","arxiv_id":"2501.18665","n_code_links":0,"syntology":null},{"paper":null,"title":"A Combined Encoder and Transformer Approach for Coherent and High-Quality Text Generation","date":"2024-11-19","arxiv_id":"2411.12157","n_code_links":0,"syntology":null},{"paper":null,"title":"No Argument Left Behind: Overlapping Chunks for Faster Processing of Arbitrarily Long Legal Texts","date":"2024-10-24","arxiv_id":"2410.19184","n_code_links":0,"syntology":null},{"paper":null,"title":"Large Body Language Models","date":"2024-10-21","arxiv_id":"2410.16533","n_code_links":0,"syntology":null},{"paper":"/paper/rico-reddit-ideological-communities","title":"RICo: Reddit ideological communities","date":"2024-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Transformers for Supervised Online Continual Learning","date":"2024-03-03","arxiv_id":"2403.01554","n_code_links":0,"syntology":null},{"paper":"/paper/unimem-towards-a-unified-view-of-long-context","title":"UniMem: Towards a Unified View of Long-Context Large Language Models","date":"2024-02-05","arxiv_id":"2402.03009","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-multi-level-threats-in-telegram","title":"Exploring Multi-Level Threats in Telegram Data with AI-Human Annotation: A Preliminary Study","date":"2023-12-15","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Illicit Darkweb Classification via Natural-language Processing: Classifying Illicit Content of Webpages based on Textual Information","date":"2023-12-08","arxiv_id":"2312.04944","n_code_links":0,"syntology":null},{"paper":"/paper/memory-efficient-stochastic-methods-for","title":"Memory-efficient Stochastic methods for Memory-based Transformers","date":"2023-11-14","arxiv_id":"2311.08123","n_code_links":1,"syntology":null},{"paper":"/paper/trams-training-free-memory-selection-for-long","title":"TRAMS: Training-free Memory Selection for Long-range Language Modeling","date":"2023-10-24","arxiv_id":"2310.15494","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/approximating-two-layer-feedforward-networks","title":"Approximating Two-Layer Feedforward Networks for Efficient Transformers","date":"2023-10-16","arxiv_id":"2310.10837","n_code_links":2,"syntology":{"ran":3,"of":4,"unverified":1,"pointer_only":0}},{"paper":"/paper/memory-gym-partially-observable-challenges-to","title":"Memory Gym: Towards Endless Tasks to Benchmark Memory Capabilities of Agents","date":"2023-09-29","arxiv_id":"2309.17207","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/random-access-infinite-context-length-for","title":"Random-Access Infinite Context Length for Transformers","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/rcmha-relative-convolutional-multi-head","title":"RCMHA: Relative Convolutional Multi-Head Attention for Natural Language Modelling","date":"2023-08-07","arxiv_id":"2308.03429","n_code_links":1,"syntology":null},{"paper":"/paper/landmark-attention-random-access-infinite","title":"Landmark Attention: Random-Access Infinite Context Length for Transformers","date":"2023-05-25","arxiv_id":"2305.16300","n_code_links":2,"syntology":{"ran":11,"of":13,"unverified":2,"pointer_only":0}},{"paper":null,"title":"Sparsified Model Zoo Twins: Investigating Populations of Sparsified Neural Network Models","date":"2023-04-26","arxiv_id":"2304.13718","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","n_code_links":1,"syntology":{"ran":16,"of":25,"unverified":9,"pointer_only":0}},{"paper":null,"title":"GTR-CTRL: Instrument and Genre Conditioning for Guitar-Focused Music Generation with Transformers","date":"2023-02-10","arxiv_id":"2302.05393","n_code_links":0,"syntology":null},{"paper":null,"title":"An Comparative Analysis of Different Pitch and Metrical Grid Encoding Methods in the Task of Sequential Music Generation","date":"2023-01-31","arxiv_id":"2301.13383","n_code_links":0,"syntology":null},{"paper":null,"title":"Efficient Sparsely Activated Transformers","date":"2022-08-31","arxiv_id":"2208.14580","n_code_links":0,"syntology":null},{"paper":"/paper/adan-adaptive-nesterov-momentum-algorithm-for","title":"Adan: Adaptive Nesterov Momentum Algorithm for Faster Optimizing Deep Models","date":"2022-08-13","arxiv_id":"2208.06677","n_code_links":9,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/recurrent-memory-transformer","title":"Recurrent Memory Transformer","date":"2022-07-14","arxiv_id":"2207.06881","n_code_links":3,"syntology":{"ran":6,"of":13,"unverified":7,"pointer_only":2}},{"paper":null,"title":"Explainable and High-Performance Hate and Offensive Speech Detection","date":"2022-06-26","arxiv_id":"2206.12983","n_code_links":0,"syntology":null},{"paper":"/paper/emotion-aware-transformer-encoder-for-1","title":"Emotion-Aware Transformer Encoder for Empathetic Dialogue Generation","date":"2022-04-24","arxiv_id":"2204.11320","n_code_links":1,"syntology":null},{"paper":"/paper/sintra-learning-an-inspiration-model-from-a","title":"SinTra: Learning an inspiration model from a single multi-track music segment","date":"2022-04-21","arxiv_id":"2204.09917","n_code_links":1,"syntology":null},{"paper":"/paper/litetransformersearch-training-free-on-device","title":"LiteTransformerSearch: Training-free Neural Architecture Search for Efficient Language Models","date":"2022-03-04","arxiv_id":"2203.02094","n_code_links":1,"syntology":{"ran":1,"of":5,"unverified":4,"pointer_only":0}},{"paper":null,"title":"Reconsidering the Past: Optimizing Hidden States in Language Models","date":"2021-12-16","arxiv_id":"2112.08653","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":62},{"task":"/task/language-modeling","name":"Language Modeling","papers":46},{"task":"/task/text-classification","name":"Text Classification","papers":17},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":16},{"task":"/task/classification","name":"General Classification","papers":15},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":13},{"task":"/task/text-classification-1","name":"text-classification","papers":12},{"task":"/task/decoder","name":"Decoder","papers":9},{"task":"/task/classification-1","name":"Classification","papers":8},{"task":"/task/machine-translation","name":"Machine Translation","papers":8},{"task":"/task/translation","name":"Translation","papers":8},{"task":"/task/speech-recognition","name":"Speech Recognition","papers":7},{"task":"/task/sentence","name":"Sentence","papers":6},{"task":"/task/word-embeddings","name":"Word Embeddings","papers":6},{"task":"/task/speech-recognition-1","name":"speech-recognition","papers":6},{"task":"/task/image-classification","name":"Image Classification","papers":5},{"task":"/task/variational-inference","name":"Variational Inference","papers":5},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":4},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":4},{"task":"/task/decision-making","name":"Decision Making","papers":4}],"tasks_shown":20,"n_tasks":144,"usage_by_year":[{"year":"2015","papers":1},{"year":"2016","papers":2},{"year":"2017","papers":7},{"year":"2018","papers":6},{"year":"2019","papers":29},{"year":"2020","papers":35},{"year":"2021","papers":32},{"year":"2022","papers":7},{"year":"2023","papers":13},{"year":"2024","papers":6},{"year":"2025","papers":3}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/variational-dropout"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}