{"url":"/method/adaptive-softmax","slug":"adaptive-softmax","name":"Adaptive Softmax","full_name":"Adaptive Softmax","full_name_withheld":false,"description_markdown":"**Adaptive Softmax** is a speedup technique for the computation of probability distributions over words. The adaptive [softmax](https://paperswithcode.com/method/softmax) is inspired by the class-based [hierarchical softmax](https://paperswithcode.com/method/hierarchical-softmax), where the word classes are built to minimize the computation time. Adaptive softmax achieves efficiency by explicitly taking into account the computation time of matrix-multiplication on parallel systems and combining it with a few important observations, namely keeping a shortlist of frequent words in the root node\r\nand reducing the capacity of rare words.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Efficient softmax approximation for GPUs","paper":"/paper/efficient-softmax-approximation-for-gpus","first_author":"Edouard Grave","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/efficient-softmax-approximation-for-gpus"},"source":{"url":"http://arxiv.org/abs/1609.04309v3","title":"Efficient softmax approximation for GPUs","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/rosinality/adaptive-softmax-pytorch/blob/2f02ca68d34f83867bea385c4be23a5392823f1e/adasoft.py#L6","code_snippet_url_on_a_code_host":true,"categories":[{"area":"General","area_id":"general","collection":"Output Functions","url":"/methods/category/output-functions","pwc_aliases":[]}],"n_papers_tagged":72,"archive_num_papers":72,"papers_newest_first":[{"paper":"/paper/rlbenchnet-the-right-network-for-the-right","title":"RLBenchNet: The Right Network for the Right Reinforcement Learning Task","date":"2025-05-21","arxiv_id":"2505.15040","n_code_links":1,"syntology":null},{"paper":null,"title":"VQ-Logits: Compressing the Output Bottleneck of Large Language Models via Vector Quantized Logits","date":"2025-05-15","arxiv_id":"2505.10202","n_code_links":0,"syntology":null},{"paper":null,"title":"Convergence Rates for Softmax Gating Mixture of Experts","date":"2025-03-05","arxiv_id":"2503.03213","n_code_links":0,"syntology":null},{"paper":null,"title":"A Combined Encoder and Transformer Approach for Coherent and High-Quality Text Generation","date":"2024-11-19","arxiv_id":"2411.12157","n_code_links":0,"syntology":null},{"paper":null,"title":"Large Body Language Models","date":"2024-10-21","arxiv_id":"2410.16533","n_code_links":0,"syntology":null},{"paper":"/paper/denomamba-a-fused-state-space-model-for-low","title":"DenoMamba: A fused state-space model for low-dose CT denoising","date":"2024-09-19","arxiv_id":"2409.13094","n_code_links":1,"syntology":null},{"paper":null,"title":"Online Residual Learning from Offline Experts for Pedestrian Tracking","date":"2024-09-06","arxiv_id":"2409.04069","n_code_links":0,"syntology":null},{"paper":null,"title":"Transformers for Supervised Online Continual Learning","date":"2024-03-03","arxiv_id":"2403.01554","n_code_links":0,"syntology":null},{"paper":"/paper/unimem-towards-a-unified-view-of-long-context","title":"UniMem: Towards a Unified View of Long-Context Large Language Models","date":"2024-02-05","arxiv_id":"2402.03009","n_code_links":1,"syntology":null},{"paper":"/paper/memory-efficient-stochastic-methods-for","title":"Memory-efficient Stochastic methods for Memory-based Transformers","date":"2023-11-14","arxiv_id":"2311.08123","n_code_links":1,"syntology":null},{"paper":"/paper/trams-training-free-memory-selection-for-long","title":"TRAMS: Training-free Memory Selection for Long-range Language Modeling","date":"2023-10-24","arxiv_id":"2310.15494","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/approximating-two-layer-feedforward-networks","title":"Approximating Two-Layer Feedforward Networks for Efficient Transformers","date":"2023-10-16","arxiv_id":"2310.10837","n_code_links":2,"syntology":{"ran":3,"of":4,"unverified":1,"pointer_only":0}},{"paper":"/paper/memory-gym-partially-observable-challenges-to","title":"Memory Gym: Towards Endless Tasks to Benchmark Memory Capabilities of Agents","date":"2023-09-29","arxiv_id":"2309.17207","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/random-access-infinite-context-length-for","title":"Random-Access Infinite Context Length for Transformers","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/rcmha-relative-convolutional-multi-head","title":"RCMHA: Relative Convolutional Multi-Head Attention for Natural Language Modelling","date":"2023-08-07","arxiv_id":"2308.03429","n_code_links":1,"syntology":null},{"paper":"/paper/landmark-attention-random-access-infinite","title":"Landmark Attention: Random-Access Infinite Context Length for Transformers","date":"2023-05-25","arxiv_id":"2305.16300","n_code_links":2,"syntology":{"ran":11,"of":13,"unverified":2,"pointer_only":0}},{"paper":"/paper/transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","n_code_links":1,"syntology":{"ran":16,"of":25,"unverified":9,"pointer_only":0}},{"paper":null,"title":"GTR-CTRL: Instrument and Genre Conditioning for Guitar-Focused Music Generation with Transformers","date":"2023-02-10","arxiv_id":"2302.05393","n_code_links":0,"syntology":null},{"paper":null,"title":"An Comparative Analysis of Different Pitch and Metrical Grid Encoding Methods in the Task of Sequential Music Generation","date":"2023-01-31","arxiv_id":"2301.13383","n_code_links":0,"syntology":null},{"paper":null,"title":"Efficient Sparsely Activated Transformers","date":"2022-08-31","arxiv_id":"2208.14580","n_code_links":0,"syntology":null},{"paper":"/paper/adan-adaptive-nesterov-momentum-algorithm-for","title":"Adan: Adaptive Nesterov Momentum Algorithm for Faster Optimizing Deep Models","date":"2022-08-13","arxiv_id":"2208.06677","n_code_links":9,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/recurrent-memory-transformer","title":"Recurrent Memory Transformer","date":"2022-07-14","arxiv_id":"2207.06881","n_code_links":3,"syntology":{"ran":6,"of":13,"unverified":7,"pointer_only":2}},{"paper":"/paper/emotion-aware-transformer-encoder-for-1","title":"Emotion-Aware Transformer Encoder for Empathetic Dialogue Generation","date":"2022-04-24","arxiv_id":"2204.11320","n_code_links":1,"syntology":null},{"paper":"/paper/sintra-learning-an-inspiration-model-from-a","title":"SinTra: Learning an inspiration model from a single multi-track music segment","date":"2022-04-21","arxiv_id":"2204.09917","n_code_links":1,"syntology":null},{"paper":"/paper/litetransformersearch-training-free-on-device","title":"LiteTransformerSearch: Training-free Neural Architecture Search for Efficient Language Models","date":"2022-03-04","arxiv_id":"2203.02094","n_code_links":1,"syntology":{"ran":1,"of":5,"unverified":4,"pointer_only":0}},{"paper":null,"title":"Reconsidering the Past: Optimizing Hidden States in Language Models","date":"2021-12-16","arxiv_id":"2112.08653","n_code_links":0,"syntology":null},{"paper":null,"title":"A Comparative Study of Transformers on Word Sense Disambiguation","date":"2021-11-30","arxiv_id":"2111.15417","n_code_links":0,"syntology":null},{"paper":null,"title":"How much do language models copy from their training data? Evaluating linguistic novelty in text generation using RAVEN","date":"2021-11-18","arxiv_id":"2111.09509","n_code_links":0,"syntology":null},{"paper":"/paper/synthesizing-collective-communication","title":"TACCL: Guiding Collective Algorithm Synthesis using Communication Sketches","date":"2021-11-08","arxiv_id":"2111.04867","n_code_links":2,"syntology":null},{"paper":null,"title":"Language Modelling via Learning to Rank","date":"2021-10-13","arxiv_id":"2110.06961","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":42},{"task":"/task/language-modeling","name":"Language Modeling","papers":31},{"task":"/task/decoder","name":"Decoder","papers":7},{"task":"/task/machine-translation","name":"Machine Translation","papers":7},{"task":"/task/translation","name":"Translation","papers":6},{"task":"/task/speech-recognition","name":"Speech Recognition","papers":5},{"task":"/task/speech-recognition-1","name":"speech-recognition","papers":5},{"task":"/task/sentence","name":"Sentence","papers":4},{"task":"/task/text-generation","name":"Text Generation","papers":4},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":3},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":3},{"task":"/task/paraphrase-identification","name":"Paraphrase Identification","papers":3},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":3},{"task":"/task/word-embeddings","name":"Word Embeddings","papers":3},{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":2},{"task":"/task/deep-attention","name":"Deep Attention","papers":2},{"task":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning","papers":2},{"task":null,"name":"GPU","papers":2},{"task":"/task/graph-neural-network","name":"Graph Neural Network","papers":2},{"task":"/task/music-generation","name":"Music Generation","papers":2}],"tasks_shown":20,"n_tasks":96,"usage_by_year":[{"year":"2016","papers":2},{"year":"2018","papers":1},{"year":"2019","papers":13},{"year":"2020","papers":17},{"year":"2021","papers":14},{"year":"2022","papers":6},{"year":"2023","papers":10},{"year":"2024","papers":6},{"year":"2025","papers":3}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/adaptive-softmax"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}