{"url":"/method/ul2","slug":"ul2","name":"UL2","full_name":"UL2","full_name_withheld":false,"description_markdown":"**UL2** is a unified framework for pretraining models that are universally effective across datasets and setups. UL2 uses Mixture-of-Denoisers (MoD), a pre-training objective that combines diverse pre-training paradigms together. UL2 introduces a notion of mode switching, wherein downstream fine-tuning is associated with specific pre-training schemes.","description_state":"present","introduced_year":null,"introduced_by":{"title":"UL2: Unifying Language Learning Paradigms","paper":"/paper/unifying-language-learning-paradigms","first_author":"Yi Tay","n_authors":14,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/unifying-language-learning-paradigms"},"source":{"url":"https://arxiv.org/abs/2205.05131v3","title":"UL2: Unifying Language Learning Paradigms","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":8,"archive_num_papers":8,"papers_newest_first":[{"paper":null,"title":"Efficient Stagewise Pretraining via Progressive Subnetworks","date":"2024-02-08","arxiv_id":"2402.05913","n_code_links":0,"syntology":null},{"paper":"/paper/turna-a-turkish-encoder-decoder-language","title":"TURNA: A Turkish Encoder-Decoder Language Model for Enhanced Understanding and Generation","date":"2024-01-25","arxiv_id":"2401.14373","n_code_links":2,"syntology":null},{"paper":null,"title":"Towards leveraging LLMs for Conditional QA","date":"2023-12-02","arxiv_id":"2312.01143","n_code_links":0,"syntology":null},{"paper":null,"title":"A Zero-shot and Few-shot Study of Instruction-Finetuned Large Language Models Applied to Clinical and Biomedical Tasks","date":"2023-07-22","arxiv_id":"2307.12114","n_code_links":0,"syntology":null},{"paper":"/paper/mlongt5-a-multilingual-and-efficient-text-to","title":"mLongT5: A Multilingual and Efficient Text-To-Text Transformer for Longer Sequences","date":"2023-05-18","arxiv_id":"2305.11129","n_code_links":1,"syntology":null},{"paper":null,"title":"ImPaKT: A Dataset for Open-Schema Knowledge Base Construction","date":"2022-12-21","arxiv_id":"2212.10770","n_code_links":0,"syntology":null},{"paper":"/paper/recitation-augmented-language-models","title":"Recitation-Augmented Language Models","date":"2022-10-04","arxiv_id":"2210.01296","n_code_links":1,"syntology":{"ran":1,"of":3,"unverified":2,"pointer_only":3}},{"paper":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","arxiv_id":"2205.05131","n_code_links":2,"syntology":{"ran":0,"of":16,"unverified":16,"pointer_only":0}}],"papers_shown":8,"tasks":[{"task":"/task/question-answering","name":"Question Answering","papers":5},{"task":"/task/retrieval","name":"Retrieval","papers":3},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/language-modelling","name":"Language Modelling","papers":2},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":2},{"task":"/task/relation-extraction","name":"Relation Extraction","papers":2},{"task":"/task/arithmetic-reasoning","name":"Arithmetic Reasoning","papers":1},{"task":"/task/attribute","name":"Attribute","papers":1},{"task":"/task/common-sense-reasoning","name":"Common Sense Reasoning","papers":1},{"task":"/task/coreference-resolution","name":"Coreference Resolution","papers":1},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/extractive-question-answering","name":"Extractive Question-Answering","papers":1},{"task":"/task/in-context-learning","name":"In-Context Learning","papers":1},{"task":"/task/inductive-bias","name":"Inductive Bias","papers":1},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":1},{"task":"/task/knowledge-base-construction","name":"Knowledge Base Construction","papers":1},{"task":"/task/long-range-modeling","name":"Long-range modeling","papers":1},{"task":"/task/mmlu","name":"MMLU","papers":1},{"task":"/task/multi-task-language-understanding","name":"Multi-task Language Understanding","papers":1},{"task":"/task/cg","name":"NER","papers":1}],"tasks_shown":20,"n_tasks":30,"usage_by_year":[{"year":"2022","papers":3},{"year":"2023","papers":3},{"year":"2024","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/ul2"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}