{"url":"/method/flan-t5","slug":"flan-t5","name":"Flan-T5","full_name":"Flan-T5","full_name_withheld":false,"description_markdown":"**Flan-T5** is the instruction fine-tuned version of **T5** or **Text-to-Text Transfer Transformer** Language Model.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Scaling Instruction-Finetuned Language Models","paper":"/paper/scaling-instruction-finetuned-language-models","first_author":"Hyung Won Chung","n_authors":35,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/scaling-instruction-finetuned-language-models"},"source":{"url":"https://arxiv.org/abs/2210.11416v5","title":"Scaling Instruction-Finetuned Language Models","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":107,"archive_num_papers":107,"papers_newest_first":[{"paper":null,"title":"LiLM-RDB-SFC: Lightweight Language Model with Relational Database-Guided DRL for Optimized SFC Provisioning","date":"2025-07-15","arxiv_id":"2507.10903","n_code_links":0,"syntology":null},{"paper":"/paper/value-free-policy-optimization-via-reward","title":"Value-Free Policy Optimization via Reward Partitioning","date":"2025-06-16","arxiv_id":"2506.13702","n_code_links":1,"syntology":null},{"paper":"/paper/masking-in-multi-hop-qa-an-analysis-of-how","title":"Masking in Multi-hop QA: An Analysis of How Language Models Perform with Context Permutation","date":"2025-05-16","arxiv_id":"2505.11754","n_code_links":1,"syntology":null},{"paper":"/paper/advancing-email-spam-detection-leveraging","title":"Advancing Email Spam Detection: Leveraging Zero-Shot Learning and Large Language Models","date":"2025-05-05","arxiv_id":"2505.02362","n_code_links":1,"syntology":null},{"paper":"/paper/ip-crr-information-pursuit-for-interpretable","title":"IP-CRR: Information Pursuit for Interpretable Classification of Chest Radiology Reports","date":"2025-04-30","arxiv_id":"2505.00191","n_code_links":1,"syntology":null},{"paper":null,"title":"Leveraging MoE-based Large Language Model for Zero-Shot Multi-Task Semantic Communication","date":"2025-03-19","arxiv_id":"2503.15722","n_code_links":0,"syntology":null},{"paper":null,"title":"MoEMoE: Question Guided Dense and Scalable Sparse Mixture-of-Expert for Multi-source Multi-modal Answering","date":"2025-03-08","arxiv_id":"2503.06296","n_code_links":0,"syntology":null},{"paper":null,"title":"Pay Attention to Real World Perturbations! Natural Robustness Evaluation in Machine Reading Comprehension","date":"2025-02-23","arxiv_id":"2502.16523","n_code_links":0,"syntology":null},{"paper":"/paper/star-spectral-truncation-and-rescale-for","title":"STAR: Spectral Truncation and Rescale for Model Merging","date":"2025-02-14","arxiv_id":"2502.10339","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":null,"title":"Cross-Language Approach for Quranic QA","date":"2025-01-29","arxiv_id":"2501.17449","n_code_links":0,"syntology":null},{"paper":"/paper/flanec-exploring-flan-t5-for-post-asr-error","title":"FlanEC: Exploring Flan-T5 for Post-ASR Error Correction","date":"2025-01-22","arxiv_id":"2501.12979","n_code_links":1,"syntology":null},{"paper":"/paper/label-privacy-in-split-learning-for-large","title":"Label Privacy in Split Learning for Large Models with Parameter-Efficient Training","date":"2024-12-21","arxiv_id":"2412.16669","n_code_links":1,"syntology":null},{"paper":null,"title":"Transforming NLU with Babylon: A Case Study in Development of Real-time, Edge-Efficient, Multi-Intent Translation System for Automated Drive-Thru Ordering","date":"2024-11-22","arxiv_id":"2411.15372","n_code_links":0,"syntology":null},{"paper":null,"title":"Reasoning or a Semblance of it? A Diagnostic Study of Transitive Reasoning in LLMs","date":"2024-10-26","arxiv_id":"2410.20200","n_code_links":0,"syntology":null},{"paper":null,"title":"Guardians of Discourse: Evaluating LLMs on Multilingual Offensive Language Detection","date":"2024-10-21","arxiv_id":"2410.15623","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhanced Electronic Health Records Text Summarization Using Large Language Models","date":"2024-10-12","arxiv_id":"2410.09628","n_code_links":0,"syntology":null},{"paper":null,"title":"Humanity in AI: Detecting the Personality of Large Language Models","date":"2024-10-11","arxiv_id":"2410.08545","n_code_links":0,"syntology":null},{"paper":null,"title":"How to Make LLMs Strong Node Classifiers?","date":"2024-10-03","arxiv_id":"2410.02296","n_code_links":0,"syntology":null},{"paper":"/paper/crowdcounter-a-benchmark-type-specific-multi","title":"CrowdCounter: A benchmark type-specific multi-target counterspeech dataset","date":"2024-10-02","arxiv_id":"2410.01400","n_code_links":1,"syntology":null},{"paper":null,"title":"Emotion-Aware Embedding Fusion in LLMs (Flan-T5, LLAMA 2, DeepSeek-R1, and ChatGPT 4) for Intelligent Response Generation","date":"2024-10-02","arxiv_id":"2410.01306","n_code_links":0,"syntology":null},{"paper":"/paper/taskcomplexity-a-dataset-for-task-complexity","title":"TaskComplexity: A Dataset for Task Complexity Classification with In-Context Learning, FLAN-T5 and GPT-4o Benchmarks","date":"2024-09-30","arxiv_id":"2409.20189","n_code_links":1,"syntology":null},{"paper":null,"title":"MIMII-Gen: Generative Modeling Approach for Simulated Evaluation of Anomalous Sound Detection System","date":"2024-09-27","arxiv_id":"2409.18542","n_code_links":0,"syntology":null},{"paper":null,"title":"Unsupervised Text Representation Learning via Instruction-Tuning for Zero-Shot Dense Retrieval","date":"2024-09-24","arxiv_id":"2409.16497","n_code_links":0,"syntology":null},{"paper":"/paper/cacer-clinical-concept-annotations-for-cancer","title":"CACER: Clinical Concept Annotations for Cancer Events and Relations","date":"2024-09-05","arxiv_id":"2409.03905","n_code_links":1,"syntology":null},{"paper":null,"title":"Instruction Finetuning for Leaderboard Generation from Empirical AI Research","date":"2024-08-19","arxiv_id":"2408.10141","n_code_links":0,"syntology":null},{"paper":"/paper/smile-zero-shot-sparse-mixture-of-low-rank","title":"SMILE: Zero-Shot Sparse Mixture of Low-Rank Experts Construction From Pre-Trained Foundation Models","date":"2024-08-19","arxiv_id":"2408.10174","n_code_links":1,"syntology":null},{"paper":"/paper/towards-enhancing-coherence-in-extractive","title":"Towards Enhancing Coherence in Extractive Summarization: Dataset and Experiments with LLMs","date":"2024-07-05","arxiv_id":"2407.04855","n_code_links":1,"syntology":null},{"paper":null,"title":"Convolutional vs Large Language Models for Software Log Classification in Edge-Deployable Cellular Network Testing","date":"2024-07-04","arxiv_id":"2407.03759","n_code_links":0,"syntology":null},{"paper":"/paper/deep-content-understanding-toward-entity-and","title":"Deep Content Understanding Toward Entity and Aspect Target Sentiment Analysis on Foundation Models","date":"2024-07-04","arxiv_id":"2407.04050","n_code_links":1,"syntology":null},{"paper":null,"title":"Factual Dialogue Summarization via Learning from Large Language Models","date":"2024-06-20","arxiv_id":"2406.14709","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":22},{"task":"/task/question-answering","name":"Question Answering","papers":19},{"task":"/task/language-modeling","name":"Language Modeling","papers":18},{"task":"/task/large-language-model","name":"Large Language Model","papers":10},{"task":"/task/in-context-learning","name":"In-Context Learning","papers":9},{"task":"/task/retrieval","name":"Retrieval","papers":8},{"task":"/task/sentence","name":"Sentence","papers":8},{"task":"/task/text-generation","name":"Text Generation","papers":7},{"task":"/task/decoder","name":"Decoder","papers":6},{"task":"/task/parameter-efficient-fine-tuning","name":"parameter-efficient fine-tuning","papers":6},{"task":"/task/benchmarking","name":"Benchmarking","papers":4},{"task":"/task/relation-extraction","name":"Relation Extraction","papers":4},{"task":"/task/zero-shot-learning","name":"Zero-Shot Learning","papers":4},{"task":"/task/attribute","name":"Attribute","papers":3},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":3},{"task":"/task/diagnostic","name":"Diagnostic","papers":3},{"task":"/task/domain-adaptation","name":"Domain Adaptation","papers":3},{"task":"/task/mmlu","name":"MMLU","papers":3},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":3},{"task":"/task/reading-comprehension","name":"Reading Comprehension","papers":3}],"tasks_shown":20,"n_tasks":167,"usage_by_year":[{"year":"2022","papers":3},{"year":"2023","papers":47},{"year":"2024","papers":46},{"year":"2025","papers":11}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/flan-t5"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}