{"url":"/method/universal-transformer","slug":"universal-transformer","name":"Universal Transformer","full_name":"Universal Transformer","full_name_withheld":false,"description_markdown":"The **Universal Transformer** is a generalization of the [Transformer](https://paperswithcode.com/method/transformer) architecture. Universal Transformers combine the parallelizability and global receptive field of feed-forward sequence models like the Transformer with the recurrent inductive bias of [RNNs](https://paperswithcode.com/methods/category/recurrent-neural-networks). They also utilise a dynamic per-position halting mechanism.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Universal Transformers","paper":"/paper/universal-transformers","first_author":"Mostafa Dehghani","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/universal-transformers"},"source":{"url":"http://arxiv.org/abs/1807.03819v3","title":"Universal Transformers","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Autoregressive Transformers","url":"/methods/category/autoregressive-transformers","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":18,"archive_num_papers":18,"papers_newest_first":[{"paper":null,"title":"PLUTO: Pathology-Universal Transformer","date":"2024-05-13","arxiv_id":"2405.07905","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-transformers-with-dynamic-halt","title":"Investigating Recurrent Transformers with Dynamic Halt","date":"2024-02-01","arxiv_id":"2402.00976","n_code_links":1,"syntology":null},{"paper":null,"title":"Self-Critical Alternate Learning based Semantic Broadcast Communication","date":"2023-12-03","arxiv_id":"2312.01423","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-universal-transformer","title":"Sparse Universal Transformer","date":"2023-10-11","arxiv_id":"2310.07096","n_code_links":2,"syntology":{"ran":3,"of":5,"unverified":2,"pointer_only":0}},{"paper":"/paper/ummaformer-a-universal-multimodal-adaptive-1","title":"UMMAFormer: A Universal Multimodal-adaptive Transformer Framework for Temporal Forgery Localization","date":"2023-08-28","arxiv_id":"2308.14395","n_code_links":1,"syntology":null},{"paper":"/paper/beyond-universal-transformer-block-reusing","title":"Beyond Universal Transformer: block reusing with adaptor in Transformer for automatic speech recognition","date":"2023-03-23","arxiv_id":"2303.13072","n_code_links":0,"syntology":null},{"paper":null,"title":"Semantic Communication with Memory","date":"2023-03-22","arxiv_id":"2303.12335","n_code_links":0,"syntology":null},{"paper":"/paper/towards-autoformalization-of-mathematics-and","title":"Towards Autoformalization of Mathematics and Code Correctness: Experiments with Elementary Proofs","date":"2023-01-05","arxiv_id":"2301.02195","n_code_links":1,"syntology":{"ran":0,"of":7,"unverified":7,"pointer_only":0}},{"paper":null,"title":"Universal Transformer Hawkes Process with Adaptive Recursive Iteration","date":"2021-12-29","arxiv_id":"2112.14479","n_code_links":0,"syntology":null},{"paper":"/paper/the-devil-is-in-the-detail-simple-tricks","title":"The Devil is in the Detail: Simple Tricks Improve Systematic Generalization of Transformers","date":"2021-08-26","arxiv_id":"2108.12284","n_code_links":2,"syntology":{"ran":0,"of":2,"unverified":2,"pointer_only":0}},{"paper":null,"title":"Using BERT Encoding and Sentence-Level Language Model for Sentence Ordering","date":"2021-08-24","arxiv_id":"2108.10986","n_code_links":0,"syntology":null},{"paper":null,"title":"Semantic Communication with Adaptive Universal Transformer","date":"2021-08-20","arxiv_id":"2108.09119","n_code_links":0,"syntology":null},{"paper":null,"title":"Automatically Ranked Russian Paraphrase Corpus for Text Generation","date":"2020-06-17","arxiv_id":"2006.09719","n_code_links":0,"syntology":null},{"paper":"/paper/universal-transforming-geometric-network","title":"Universal Transforming Geometric Network","date":"2019-08-02","arxiv_id":"1908.00723","n_code_links":1,"syntology":null},{"paper":null,"title":"Latent Universal Task-Specific BERT","date":"2019-05-16","arxiv_id":"1905.06638","n_code_links":0,"syntology":null},{"paper":null,"title":"Self-Attentive Model for Headline Generation","date":"2019-01-23","arxiv_id":"1901.07786","n_code_links":0,"syntology":null},{"paper":"/paper/attending-to-mathematical-language-with","title":"Attending to Mathematical Language with Transformers","date":"2018-12-05","arxiv_id":"1812.02825","n_code_links":3,"syntology":null},{"paper":"/paper/universal-transformers","title":"Universal Transformers","date":"2018-07-10","arxiv_id":"1807.03819","n_code_links":8,"syntology":{"ran":14,"of":25,"unverified":11,"pointer_only":24}}],"papers_shown":18,"tasks":[{"task":"/task/sentence","name":"Sentence","papers":4},{"task":"/task/language-modeling","name":"Language Modeling","papers":3},{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":"/task/semantic-communication","name":"Semantic Communication","papers":3},{"task":"/task/sentence-similarity","name":"Sentence Similarity","papers":2},{"task":"/task/text-generation","name":"Text Generation","papers":2},{"task":"/task/articles","name":"Articles","papers":1},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":1},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":1},{"task":"/task/binary-classification","name":"Binary Classification","papers":1},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/diagnostic","name":"Diagnostic","papers":1},{"task":"/task/document-summarization","name":"Document Summarization","papers":1},{"task":"/task/headline-generation","name":"Headline Generation","papers":1},{"task":"/task/inductive-bias","name":"Inductive Bias","papers":1},{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":1},{"task":"/task/lambada","name":"LAMBADA","papers":1},{"task":"/task/learning-to-execute","name":"Learning to Execute","papers":1},{"task":"/task/listops","name":"ListOps","papers":1},{"task":"/task/logical-sequence","name":"Logical Sequence","papers":1}],"tasks_shown":20,"n_tasks":43,"usage_by_year":[{"year":"2018","papers":2},{"year":"2019","papers":3},{"year":"2020","papers":1},{"year":"2021","papers":4},{"year":"2023","papers":6},{"year":"2024","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/universal-transformer"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}