{"url":"/method/mbert","slug":"mbert","name":"mBERT","full_name":"mBERT","full_name_withheld":false,"description_markdown":"mBERT","description_state":"present","introduced_year":null,"introduced_by":{"title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","paper":"/paper/bert-pre-training-of-deep-bidirectional","first_author":"Jacob Devlin","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/bert-pre-training-of-deep-bidirectional"},"source":{"url":"https://arxiv.org/abs/1810.04805v2","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":198,"archive_num_papers":198,"papers_newest_first":[{"paper":null,"title":"How do datasets, developers, and models affect biases in a low-resourced language?","date":"2025-06-07","arxiv_id":"2506.06816","n_code_links":0,"syntology":null},{"paper":null,"title":"University of Indonesia at SemEval-2025 Task 11: Evaluating State-of-the-Art Encoders for Multi-Label Emotion Detection","date":"2025-05-22","arxiv_id":"2505.16460","n_code_links":0,"syntology":null},{"paper":null,"title":"Cross-Linguistic Transfer in Multilingual NLP: The Role of Language Families and Morphology","date":"2025-05-20","arxiv_id":"2505.13908","n_code_links":0,"syntology":null},{"paper":null,"title":"Generating Synthetic Oracle Datasets to Analyze Noise Impact: A Study on Building Function Classification Using Tweets","date":"2025-03-28","arxiv_id":"2503.22856","n_code_links":0,"syntology":null},{"paper":null,"title":"Comparative Approaches to Sentiment Analysis Using Datasets in Major European and Arabic Languages","date":"2025-01-21","arxiv_id":"2501.12540","n_code_links":0,"syntology":null},{"paper":null,"title":"The First Multilingual Model For The Detection of Suicide Texts","date":"2024-12-20","arxiv_id":"2412.15498","n_code_links":0,"syntology":null},{"paper":null,"title":"Transformer-Based Contextualized Language Models Joint with Neural Networks for Natural Language Inference in Vietnamese","date":"2024-11-20","arxiv_id":"2411.13407","n_code_links":0,"syntology":null},{"paper":"/paper/from-n-grams-to-pre-trained-multilingual","title":"From N-grams to Pre-trained Multilingual Models For Language Identification","date":"2024-10-11","arxiv_id":"2410.08728","n_code_links":2,"syntology":null},{"paper":"/paper/motamot-a-dataset-for-revealing-the-supremacy","title":"Motamot: A Dataset for Revealing the Supremacy of Large Language Models over Transformer Models in Bengali Political Sentiment Analysis","date":"2024-07-28","arxiv_id":"2407.19528","n_code_links":1,"syntology":null},{"paper":null,"title":"On Initializing Transformers with Pre-trained Embeddings","date":"2024-07-17","arxiv_id":"2407.12514","n_code_links":0,"syntology":null},{"paper":null,"title":"Decipherment-Aware Multilingual Learning in Jointly Trained Language Models","date":"2024-06-11","arxiv_id":"2406.07231","n_code_links":0,"syntology":null},{"paper":null,"title":"Vietnamese AI Generated Text Detection","date":"2024-05-06","arxiv_id":"2405.03206","n_code_links":0,"syntology":null},{"paper":"/paper/unraveling-the-dominance-of-large-language","title":"Unraveling the Dominance of Large Language Models Over Transformer Models for Bangla Natural Language Inference: A Comprehensive Study","date":"2024-05-05","arxiv_id":"2405.02937","n_code_links":1,"syntology":null},{"paper":"/paper/uqa-corpus-for-urdu-question-answering","title":"UQA: Corpus for Urdu Question Answering","date":"2024-05-02","arxiv_id":"2405.01458","n_code_links":3,"syntology":null},{"paper":"/paper/incorporating-lexical-and-syntactic-knowledge","title":"Incorporating Lexical and Syntactic Knowledge for Unsupervised Cross-Lingual Transfer","date":"2024-04-25","arxiv_id":"2404.16627","n_code_links":1,"syntology":null},{"paper":null,"title":"Adapting Mental Health Prediction Tasks for Cross-lingual Learning via Meta-Training and In-context Learning with Large Language Model","date":"2024-04-13","arxiv_id":"2404.09045","n_code_links":0,"syntology":null},{"paper":"/paper/data-bias-according-to-bipol-men-are","title":"Data Bias According to Bipol: Men are Naturally Right and It is the Role of Women to Follow Their Lead","date":"2024-04-07","arxiv_id":"2404.04838","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-for-expansion-of-spoken","title":"Large Language Models for Expansion of Spoken Language Understanding Systems to New Languages","date":"2024-04-03","arxiv_id":"2404.02588","n_code_links":1,"syntology":null},{"paper":null,"title":"A Benchmark Evaluation of Clinical Named Entity Recognition in French","date":"2024-03-28","arxiv_id":"2403.19726","n_code_links":0,"syntology":null},{"paper":null,"title":"Introducing Syllable Tokenization for Low-resource Languages: A Case Study with Swahili","date":"2024-03-26","arxiv_id":"2406.15358","n_code_links":0,"syntology":null},{"paper":"/paper/ax-to-grind-urdu-benchmark-dataset-for-urdu","title":"Ax-to-Grind Urdu: Benchmark Dataset for Urdu Fake News Detection","date":"2024-03-20","arxiv_id":"2403.14037","n_code_links":1,"syntology":null},{"paper":"/paper/llmlingua-2-data-distillation-for-efficient","title":"LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression","date":"2024-03-19","arxiv_id":"2403.12968","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-named-entity-recognition","title":"Evaluating Named Entity Recognition: A comparative analysis of mono- and multilingual transformer models on a novel Brazilian corporate earnings call transcripts dataset","date":"2024-03-18","arxiv_id":"2403.12212","n_code_links":2,"syntology":null},{"paper":"/paper/a-measure-for-transparent-comparison-of","title":"A Measure for Transparent Comparison of Linguistic Diversity in Multilingual NLP Data Sets","date":"2024-03-06","arxiv_id":"2403.03909","n_code_links":1,"syntology":null},{"paper":null,"title":"Enhancing ESG Impact Type Identification through Early Fusion and Multilingual Models","date":"2024-02-16","arxiv_id":"2402.10772","n_code_links":0,"syntology":null},{"paper":"/paper/cross-lingual-editing-in-multilingual","title":"Cross-lingual Editing in Multilingual Language Models","date":"2024-01-19","arxiv_id":"2401.10521","n_code_links":1,"syntology":null},{"paper":null,"title":"LinguAlchemy: Fusing Typological and Geographical Elements for Unseen Language Generalization","date":"2024-01-11","arxiv_id":"2401.06034","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-large-language-models-leak-human","title":"Multilingual large language models leak human stereotypes across language boundaries","date":"2023-12-12","arxiv_id":"2312.07141","n_code_links":1,"syntology":null},{"paper":null,"title":"On Significance of Subword tokenization for Low Resource and Efficient Named Entity Recognition: A case study in Marathi","date":"2023-12-03","arxiv_id":"2312.01306","n_code_links":0,"syntology":null},{"paper":"/paper/vashantor-a-large-scale-multilingual","title":"Vashantor: A Large-scale Multilingual Benchmark Dataset for Automated Translation of Bangla Regional Dialects to Bangla Language","date":"2023-11-18","arxiv_id":"2311.11142","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/xlm-r","name":"XLM-R","papers":49},{"task":"/task/cross-lingual-transfer","name":"Cross-Lingual Transfer","papers":40},{"task":"/task/language-modelling","name":"Language Modelling","papers":29},{"task":"/task/zero-shot-cross-lingual-transfer","name":"Zero-Shot Cross-Lingual Transfer","papers":23},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":22},{"task":"/task/language-modeling","name":"Language Modeling","papers":20},{"task":"/task/cg","name":"NER","papers":19},{"task":"/task/named-entity-recognition-1","name":"Named Entity Recognition","papers":19},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":19},{"task":"/task/named-entity-recognition-ner","name":"Named Entity Recognition (NER)","papers":18},{"task":"/task/sentence","name":"Sentence","papers":18},{"task":"/task/named-entity-recognition","name":"named-entity-recognition","papers":16},{"task":"/task/pos","name":"POS","papers":15},{"task":"/task/pos-tagging","name":"POS Tagging","papers":13},{"task":"/task/text-classification","name":"Text Classification","papers":13},{"task":"/task/translation","name":"Translation","papers":12},{"task":"/task/word-embeddings","name":"Word Embeddings","papers":12},{"task":"/task/text-classification-1","name":"text-classification","papers":12},{"task":"/task/dependency-parsing","name":"Dependency Parsing","papers":11},{"task":"/task/machine-translation","name":"Machine Translation","papers":11}],"tasks_shown":20,"n_tasks":170,"usage_by_year":[{"year":"2018","papers":1},{"year":"2019","papers":2},{"year":"2020","papers":34},{"year":"2021","papers":56},{"year":"2022","papers":47},{"year":"2023","papers":31},{"year":"2024","papers":22},{"year":"2025","papers":5}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/mbert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}