{"url":"/method/mobilebert","slug":"mobilebert","name":"MobileBERT","full_name":"MobileBERT","full_name_withheld":false,"description_markdown":"**MobileBERT** is a type of inverted-bottleneck [BERT](https://paperswithcode.com/method/bert) that compresses and accelerates the popular BERT model. MobileBERT is a thin version of BERT_LARGE, while equipped with bottleneck structures and a carefully designed balance between self-attentions and feed-forward networks. To train MobileBERT, we first train a specially designed teacher model, an inverted-bottleneck incorporated BERT_LARGE model. Then, we conduct knowledge transfer from this teacher to MobileBERT. Like the original BERT, MobileBERT is task-agnostic, that is, it can be generically applied to various downstream NLP tasks via simple fine-tuning. It is trained by layer-to-layer imitating the inverted bottleneck BERT.","description_state":"present","introduced_year":null,"introduced_by":{"title":"MobileBERT: a Compact Task-Agnostic BERT for Resource-Limited Devices","paper":"/paper/mobilebert-a-compact-task-agnostic-bert-for","first_author":"Zhiqing Sun","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/mobilebert-a-compact-task-agnostic-bert-for"},"source":{"url":"https://arxiv.org/abs/2004.02984v2","title":"MobileBERT: a Compact Task-Agnostic BERT for Resource-Limited Devices","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Autoencoding Transformers","url":"/methods/category/autoencoding-transformers","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":12,"archive_num_papers":12,"papers_newest_first":[{"paper":null,"title":"Efficient Intent-Based Filtering for Multi-Party Conversations Using Knowledge Distillation from LLMs","date":"2025-03-21","arxiv_id":"2503.17336","n_code_links":0,"syntology":null},{"paper":null,"title":"FedMentalCare: Towards Privacy-Preserving Fine-Tuned LLMs to Analyze Mental Health Status Using Federated Learning Framework","date":"2025-02-27","arxiv_id":"2503.05786","n_code_links":0,"syntology":null},{"paper":null,"title":"Resource-Efficient Transformer Architecture: Optimizing Memory and Execution Time for Real-Time Applications","date":"2024-12-25","arxiv_id":"2501.00042","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-deployment-of-transformer-models-in","title":"Efficient Deployment of Transformer Models in Analog In-Memory Computing Hardware","date":"2024-11-26","arxiv_id":"2411.17367","n_code_links":1,"syntology":null},{"paper":null,"title":"On-Device Emoji Classifier Trained with GPT-based Data Augmentation for a Mobile Keyboard","date":"2024-11-06","arxiv_id":"2411.05031","n_code_links":0,"syntology":null},{"paper":"/paper/utilizing-large-language-models-to-optimize","title":"PhishLang: A Real-Time, Fully Client-Side Phishing Detection Framework Using MobileBERT","date":"2024-08-11","arxiv_id":"2408.05667","n_code_links":2,"syntology":null},{"paper":"/paper/toward-attention-based-tinyml-a-heterogeneous","title":"Toward Attention-based TinyML: A Heterogeneous Accelerated Architecture and Automated Deployment Flow","date":"2024-08-05","arxiv_id":"2408.02473","n_code_links":1,"syntology":null},{"paper":null,"title":"Quantized Transformer Language Model Implementations on Edge Devices","date":"2023-10-06","arxiv_id":"2310.03971","n_code_links":0,"syntology":null},{"paper":null,"title":"AutoDistill: an End-to-End Framework to Explore and Distill Hardware-Efficient Language Models","date":"2022-01-21","arxiv_id":"2201.08539","n_code_links":0,"syntology":null},{"paper":"/paper/character-level-hypernetworks-for-hate-speech","title":"Character-level HyperNetworks for Hate Speech Detection","date":"2021-11-11","arxiv_id":"2111.06336","n_code_links":1,"syntology":null},{"paper":"/paper/lidsnet-a-lightweight-on-device-intent","title":"LIDSNet: A Lightweight on-device Intent Detection model using Deep Siamese Network","date":"2021-10-06","arxiv_id":"2110.15717","n_code_links":0,"syntology":null},{"paper":"/paper/mobilebert-a-compact-task-agnostic-bert-for","title":"MobileBERT: a Compact Task-Agnostic BERT for Resource-Limited Devices","date":"2020-04-06","arxiv_id":"2004.02984","n_code_links":7,"syntology":null}],"papers_shown":12,"tasks":[{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":2},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":2},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":2},{"task":"/task/language-modelling","name":"Language Modelling","papers":2},{"task":"/task/privacy-preserving","name":"Privacy Preserving","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2},{"task":"/task/bayesian-optimization","name":"Bayesian Optimization","papers":1},{"task":"/task/federated-learning","name":"Federated Learning","papers":1},{"task":"/task/hate-speech-detection","name":"Hate Speech Detection","papers":1},{"task":"/task/intent-classification","name":"Intent Classification","papers":1},{"task":"/task/intent-detection","name":"Intent Detection","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":"/task/model-compression","name":"Model Compression","papers":1},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":1},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":1},{"task":"/task/architecture-search","name":"Neural Architecture Search","papers":1},{"task":"/task/phishing-website-detection","name":"Phishing Website Detection","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1}],"tasks_shown":20,"n_tasks":24,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":2},{"year":"2022","papers":1},{"year":"2023","papers":1},{"year":"2024","papers":5},{"year":"2025","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/mobilebert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}