{"url":"/method/mt5","slug":"mt5","name":"mT5","full_name":"mT5","full_name_withheld":false,"description_markdown":"**mt5** is a multilingual variant of [T5](https://paperswithcode.com/method/t5) that was pre-trained on a new Common Crawl-based dataset covering $101$ languages.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/2010.11934v3","title":"mT5: A massively multilingual pre-trained text-to-text transformer","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":104,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"Advancing Sentiment Analysis in Tamil-English Code-Mixed Texts: Challenges and Transformer-Based Solutions","date":"2025-03-30","arxiv_id":"2503.23295","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-robustness-of-multilingual-llms-on","title":"Exploring Robustness of Multilingual LLMs on Real-World Noisy Data","date":"2025-01-14","arxiv_id":"2501.08322","n_code_links":1,"syntology":null},{"paper":null,"title":"Bridging Dialects: Translating Standard Bangla to Regional Variants Using Neural Models","date":"2025-01-10","arxiv_id":"2501.05749","n_code_links":0,"syntology":null},{"paper":null,"title":"Empowering Bengali Education with AI: Solving Bengali Math Word Problems through Transformer Models","date":"2025-01-05","arxiv_id":"2501.02599","n_code_links":0,"syntology":null},{"paper":"/paper/belin-a-novel-corpus-for-bengali-religious","title":"BeliN: A Novel Corpus for Bengali Religious News Headline Generation using Contextual Feature Fusion","date":"2025-01-02","arxiv_id":"2501.01069","n_code_links":1,"syntology":null},{"paper":null,"title":"HAND: Hierarchical Attention Network for Multi-Scale Handwritten Document Recognition and Layout Analysis","date":"2024-12-25","arxiv_id":"2412.18981","n_code_links":0,"syntology":null},{"paper":null,"title":"Advancing Explainability in Neural Machine Translation: Analytical Metrics for Attention and Alignment Consistency","date":"2024-12-24","arxiv_id":"2412.18669","n_code_links":0,"syntology":null},{"paper":null,"title":"The First Multilingual Model For The Detection of Suicide Texts","date":"2024-12-20","arxiv_id":"2412.15498","n_code_links":0,"syntology":null},{"paper":null,"title":"Neural Text Normalization for Luxembourgish using Real-Life Variation Data","date":"2024-12-12","arxiv_id":"2412.09383","n_code_links":0,"syntology":null},{"paper":"/paper/sitse-sinhala-text-simplification-dataset-and","title":"SiTSE: Sinhala Text Simplification Dataset and Evaluation","date":"2024-12-02","arxiv_id":"2412.01293","n_code_links":1,"syntology":null},{"paper":null,"title":"Towards Santali Linguistic Inclusion: Building the First Santali-to-English Translation Model using mT5 Transformer and Data Augmentation","date":"2024-11-29","arxiv_id":"2411.19726","n_code_links":0,"syntology":null},{"paper":null,"title":"Unlocking Legal Knowledge: A Multilingual Dataset for Judicial Summarization in Switzerland","date":"2024-10-17","arxiv_id":"2410.13456","n_code_links":0,"syntology":null},{"paper":null,"title":"Tokenization and Morphology in Multilingual Language Models: A Comparative Analysis of mT5 and ByT5","date":"2024-10-15","arxiv_id":"2410.11627","n_code_links":0,"syntology":null},{"paper":null,"title":"Abstractive Summarization of Low resourced Nepali language using Multilingual Transformers","date":"2024-09-29","arxiv_id":"2409.19566","n_code_links":0,"syntology":null},{"paper":"/paper/muri-high-quality-instruction-tuning-datasets","title":"MURI: High-Quality Instruction Tuning Datasets for Low-Resource Languages via Reverse Instructions","date":"2024-09-19","arxiv_id":"2409.12958","n_code_links":1,"syntology":null},{"paper":null,"title":"Exploring Fine-tuned Generative Models for Keyphrase Selection: A Case Study for Russian","date":"2024-09-16","arxiv_id":"2409.10640","n_code_links":0,"syntology":null},{"paper":null,"title":"On Initializing Transformers with Pre-trained Embeddings","date":"2024-07-17","arxiv_id":"2407.12514","n_code_links":0,"syntology":null},{"paper":null,"title":"Segment-Based Interactive Machine Translation for Pre-trained Models","date":"2024-07-09","arxiv_id":"2407.06990","n_code_links":0,"syntology":null},{"paper":null,"title":"Vision-Braille: An End-to-End Tool for Chinese Braille Image-to-Text Translation","date":"2024-07-08","arxiv_id":"2407.06048","n_code_links":0,"syntology":null},{"paper":null,"title":"The Model Arena for Cross-lingual Sentiment Analysis: A Comparative Study in the Era of Large Language Models","date":"2024-06-27","arxiv_id":"2406.19358","n_code_links":0,"syntology":null},{"paper":"/paper/uqa-corpus-for-urdu-question-answering","title":"UQA: Corpus for Urdu Question Answering","date":"2024-05-02","arxiv_id":"2405.01458","n_code_links":3,"syntology":null},{"paper":null,"title":"TuBA: Cross-Lingual Transferability of Backdoor Attacks in LLMs with Instruction Tuning","date":"2024-04-30","arxiv_id":"2404.19597","n_code_links":0,"syntology":null},{"paper":"/paper/indicgenbench-a-multilingual-benchmark-to","title":"IndicGenBench: A Multilingual Benchmark to Evaluate Generation Capabilities of LLMs on Indic Languages","date":"2024-04-25","arxiv_id":"2404.16816","n_code_links":1,"syntology":null},{"paper":null,"title":"HLTCOE at TREC 2023 NeuCLIR Track","date":"2024-04-11","arxiv_id":"2404.08118","n_code_links":0,"syntology":null},{"paper":null,"title":"Medical mT5: An Open-Source Multilingual Text-to-Text LLM for The Medical Domain","date":"2024-04-11","arxiv_id":"2404.07613","n_code_links":0,"syntology":null},{"paper":"/paper/data-bias-according-to-bipol-men-are","title":"Data Bias According to Bipol: Men are Naturally Right and It is the Role of Women to Follow Their Lead","date":"2024-04-07","arxiv_id":"2404.04838","n_code_links":1,"syntology":null},{"paper":"/paper/adaptive-cross-lingual-text-classification","title":"Adaptive Cross-lingual Text Classification through In-Context One-Shot Demonstrations","date":"2024-04-03","arxiv_id":"2404.02452","n_code_links":1,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":2}},{"paper":null,"title":"Transcribing Bengali Text with Regional Dialects to IPA using District Guided Tokens","date":"2024-03-26","arxiv_id":"2403.17407","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-named-entity-recognition","title":"Evaluating Named Entity Recognition: A comparative analysis of mono- and multilingual transformer models on a novel Brazilian corporate earnings call transcripts dataset","date":"2024-03-18","arxiv_id":"2403.12212","n_code_links":2,"syntology":null},{"paper":null,"title":"Basque and Spanish Counter Narrative Generation: Data Creation and Evaluation","date":"2024-03-14","arxiv_id":"2403.09159","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/translation","name":"Translation","papers":17},{"task":"/task/language-modelling","name":"Language Modelling","papers":16},{"task":"/task/machine-translation","name":"Machine Translation","papers":16},{"task":"/task/question-answering","name":"Question Answering","papers":14},{"task":"/task/language-modeling","name":"Language Modeling","papers":13},{"task":"/task/decoder","name":"Decoder","papers":12},{"task":"/task/cross-lingual-transfer","name":"Cross-Lingual Transfer","papers":11},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":11},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":9},{"task":"/task/xlm-r","name":"XLM-R","papers":9},{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":8},{"task":"/task/sentence","name":"Sentence","papers":8},{"task":"/task/diversity","name":"Diversity","papers":7},{"task":"/task/denoising","name":"Denoising","papers":5},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":5},{"task":"/task/text-generation","name":"Text Generation","papers":5},{"task":"/task/text-summarization","name":"Text Summarization","papers":5},{"task":"/task/articles","name":"Articles","papers":4},{"task":"/task/masked-language-modeling","name":"Masked Language Modeling","papers":4},{"task":"/task/multilingual-nlp","name":"Multilingual NLP","papers":4}],"tasks_shown":20,"n_tasks":133,"usage_by_year":[{"year":"2020","papers":2},{"year":"2021","papers":12},{"year":"2022","papers":34},{"year":"2023","papers":21},{"year":"2024","papers":30},{"year":"2025","papers":5}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/mt5"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}