{"url":"/method/autotinybert","slug":"autotinybert","name":"AutoTinyBERT","full_name":"AutoTinyBERT","full_name_withheld":false,"description_markdown":"**AutoTinyBERT** is a an efficient [BERT](https://paperswithcode.com/method/bert) variant found through neural architecture search. Specifically, one-shot learning is used to obtain a big Super Pretrained Language Model (SuperPLM), where the objectives of pre-training or task-agnostic BERT distillation are used.  Then, given a specific latency constraint, an evolutionary algorithm is run on the SuperPLM to search optimal architectures. Finally, we extract the corresponding sub-models based on the optimal architectures and further train these models.","description_state":"present","introduced_year":null,"introduced_by":{"title":"AutoTinyBERT: Automatic Hyper-parameter Optimization for Efficient Pre-trained Language Models","paper":"/paper/autotinybert-automatic-hyper-parameter","first_author":"Yichun Yin","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/autotinybert-automatic-hyper-parameter"},"source":{"url":"https://arxiv.org/abs/2107.13686v1","title":"AutoTinyBERT: Automatic Hyper-parameter Optimization for Efficient Pre-trained Language Models","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Autoencoding Transformers","url":"/methods/category/autoencoding-transformers","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":2,"archive_num_papers":2,"papers_newest_first":[{"paper":null,"title":"SpeedLimit: Neural Architecture Search for Quantized Transformer Models","date":"2022-09-25","arxiv_id":"2209.12127","n_code_links":0,"syntology":null},{"paper":"/paper/autotinybert-automatic-hyper-parameter","title":"AutoTinyBERT: Automatic Hyper-parameter Optimization for Efficient Pre-trained Language Models","date":"2021-07-29","arxiv_id":"2107.13686","n_code_links":1,"syntology":null}],"papers_shown":2,"tasks":[{"task":"/task/architecture-search","name":"Neural Architecture Search","papers":2},{"task":"/task/one-shot-learning","name":"One-Shot Learning","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/two","name":"Vocal Bursts Valence Prediction","papers":1}],"tasks_shown":4,"n_tasks":4,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/autotinybert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}