{"url":"/method/normformer","slug":"normformer","name":"NormFormer","full_name":"NormFormer","full_name_withheld":false,"description_markdown":"**NormFormer** is a type of [Pre-LN](https://paperswithcode.com/method/layer-normalization) transformer that adds three normalization operations to each layer: a Layer Norm after self attention, head-wise scaling of self-attention outputs, and a Layer Norm after the first [fully connected layer](https://paperswithcode.com/method/position-wise-feed-forward-layer). The modifications introduce a small number of additional learnable parameters, which provide a cost-effective way for each layer to change the magnitude of its features, and therefore the magnitude of the gradients to subsequent components.","description_state":"present","introduced_year":null,"introduced_by":{"title":"NormFormer: Improved Transformer Pretraining with Extra Normalization","paper":"/paper/normformer-improved-transformer-pretraining-1","first_author":"Sam Shleifer","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/normformer-improved-transformer-pretraining-1"},"source":{"url":"https://arxiv.org/abs/2110.09456v2","title":"NormFormer: Improved Transformer Pretraining with Extra Normalization","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":1,"archive_num_papers":1,"papers_newest_first":[{"paper":"/paper/normformer-improved-transformer-pretraining-1","title":"NormFormer: Improved Transformer Pretraining with Extra Normalization","date":"2021-10-18","arxiv_id":"2110.09456","n_code_links":1,"syntology":null}],"papers_shown":1,"tasks":[{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/masked-language-modeling","name":"Masked Language Modeling","papers":1}],"tasks_shown":3,"n_tasks":3,"usage_by_year":[{"year":"2021","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/normformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}