{"url":"/method/routing-transformer","slug":"routing-transformer","name":"Routing Transformer","full_name":"Routing Transformer","full_name_withheld":false,"description_markdown":"The **Routing Transformer** is a [Transformer](https://paperswithcode.com/method/transformer) that endows self-attention with a sparse routing module based on online k-means. Each attention module considers a clustering of the space: the current timestep only attends to context belonging to the same cluster. In other word, the current time-step query is routed to a limited number of context through its cluster assignment.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Efficient Content-Based Sparse Attention with Routing Transformers","paper":"/paper/efficient-content-based-sparse-attention-with-1","first_author":"Aurko Roy","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/efficient-content-based-sparse-attention-with-1"},"source":{"url":"https://arxiv.org/abs/2003.05997v5","title":"Efficient Content-Based Sparse Attention with Routing Transformers","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Autoregressive Transformers","url":"/methods/category/autoregressive-transformers","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":null,"title":"Hybrid Routing Transformer for Zero-Shot Learning","date":"2022-03-29","arxiv_id":"2203.15310","n_code_links":0,"syntology":null},{"paper":"/paper/hurdles-to-progress-in-long-form-question","title":"Hurdles to Progress in Long-form Question Answering","date":"2021-03-10","arxiv_id":"2103.06332","n_code_links":2,"syntology":null},{"paper":"/paper/efficient-content-based-sparse-attention-with-1","title":"Efficient Content-Based Sparse Attention with Routing Transformers","date":"2020-03-12","arxiv_id":"2003.05997","n_code_links":2,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":2}}],"papers_shown":3,"tasks":[{"task":"/task/attribute","name":"Attribute","papers":1},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/form","name":"Form","papers":1},{"task":"/task/image-generation","name":"Image Generation","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/long-form-question-answering","name":"Long Form Question Answering","papers":1},{"task":"/task/open-domain-dialog","name":"Open-Domain Dialog","papers":1},{"task":"/task/open-domain-question-answering","name":"Open-Domain Question Answering","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/text-generation","name":"Text Generation","papers":1},{"task":"/task/zero-shot-learning","name":"Zero-Shot Learning","papers":1}],"tasks_shown":12,"n_tasks":12,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":1},{"year":"2022","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/routing-transformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}