{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/efficient-softmax-approximation-for-gpus","title":"Efficient softmax approximation for GPUs","arxiv_id":"1609.04309","date":"2016-09-14","proceeding":"ICML 2017 8","authors":["Edouard Grave","Armand Joulin","Moustapha Cissé","David Grangier","Hervé Jégou"],"abstract":"We propose an approximate strategy to efficiently train neural network based\nlanguage models over very large vocabularies. Our approach, called adaptive\nsoftmax, circumvents the linear dependency on the vocabulary size by exploiting\nthe unbalanced word distribution to form clusters that explicitly minimize the\nexpectation of computation time. Our approach further reduces the computational\ntime by exploiting the specificities of modern architectures and matrix-matrix\nvector operations, making it particularly suited for graphical processing\nunits. Our experiments carried out on standard benchmarks, such as EuroParl and\nOne Billion Word, show that our approach brings a large gain in efficiency over\nstandard approximations while achieving an accuracy close to that of the full\nsoftmax. The code of our method is available at\nhttps://github.com/facebookresearch/adaptive-softmax.","url_abs":"http://arxiv.org/abs/1609.04309v3","url_pdf":"http://arxiv.org/pdf/1609.04309v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/facebookresearch/adaptive-softmax","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"torch","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/DavidWBressler/adaptivesoftmax","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/Jmkernes/PAR-Transformer-XL","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/ahmedbahaaeldin/Papers-from-Scratch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/astanway/gated-conv-nets","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/cedrickchee/pytorch-pretrained-BERT","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/jiali-ms/JLM","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/rdspring1/PyTorch_GBW_LM","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/rosinality/adaptive-softmax-pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/simon555/LM_word","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/yangsaiyong/tf-adaptive-softmax-lstm-lm","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"GPL-3.0"}},{"paper_slug":"efficient-softmax-approximation-for-gpus","repo_url":"https://github.com/huggingface/transformers","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[],"methods":[{"method_slug":"adaptive-softmax","method_name":"Adaptive Softmax"}],"datasets_introduced":[],"methods_introduced":[{"slug":"adaptive-softmax","name":"Adaptive Softmax","full_name":"Adaptive Softmax"}],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1609.04309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.04309"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huggingface/transformers","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/rdspring1/PyTorch_GBW_LM","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/simon555/LM_word","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/yangsaiyong/tf-adaptive-softmax-lstm-lm","reach":{"status":"ok","spdx":"GPL-3.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Jmkernes/PAR-Transformer-XL","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/facebookresearch/adaptive-softmax","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/DavidWBressler/adaptivesoftmax","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jiali-ms/JLM","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cedrickchee/pytorch-pretrained-BERT","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/rosinality/adaptive-softmax-pytorch","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/astanway/gated-conv-nets","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ahmedbahaaeldin/Papers-from-Scratch","reach":{"status":"ok"}}],"summary":{"unverified":1},"by_repo_kind":{"listed":{"samples":1,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"16460bb9b7b5d0a1","entry":"repackage_hidden","repo":"rosinality/adaptive-softmax-pytorch","repo_kind":"listed","path":"text8.py","file_url":"https://github.com/rosinality/adaptive-softmax-pytorch/blob/HEAD/text8.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"16460bb9b7b5d0a1"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}