{"url":"/method/subformer","slug":"subformer","name":"Subformer","full_name":"Subformer","full_name_withheld":false,"description_markdown":"**Subformer** is a [Transformer](https://paperswithcode.com/method/transformer) that combines sandwich-style parameter sharing, which overcomes naive cross-layer parameter sharing in generative models, and self-attentive embedding factorization (SAFE). In SAFE, a small self-attention layer is used to reduce embedding parameter count.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Subformer: Exploring Weight Sharing for Parameter Efficiency in Generative Transformers","paper":"/paper/subformer-exploring-weight-sharing-for","first_author":"Machel Reid","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/subformer-exploring-weight-sharing-for"},"source":{"url":"https://arxiv.org/abs/2101.00234v3","title":"Subformer: Exploring Weight Sharing for Parameter Efficiency in Generative Transformers","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":"/paper/transformers-are-efficient-hierarchical","title":"Transformers are efficient hierarchical chemical graph learners","date":"2023-10-02","arxiv_id":"2310.01704","n_code_links":1,"syntology":null},{"paper":"/paper/subformer-a-parameter-reduced-transformer","title":"Subformer: A Parameter Reduced Transformer","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/subformer-exploring-weight-sharing-for","title":"Subformer: Exploring Weight Sharing for Parameter Efficiency in Generative Transformers","date":"2021-01-01","arxiv_id":"2101.00234","n_code_links":1,"syntology":null}],"papers_shown":3,"tasks":[{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":2},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/language-modelling","name":"Language Modelling","papers":2},{"task":"/task/machine-translation","name":"Machine Translation","papers":2},{"task":"/task/translation","name":"Translation","papers":2},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/graph-representation-learning","name":"Graph Representation Learning","papers":1},{"task":"/task/representation-learning","name":"Representation Learning","papers":1}],"tasks_shown":8,"n_tasks":8,"usage_by_year":[{"year":"2021","papers":2},{"year":"2023","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/subformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}