{"url":"/method/deebert","slug":"deebert","name":"DeeBERT","full_name":"DeeBERT","full_name_withheld":false,"description_markdown":"**DeeBERT** is a method for accelerating [BERT](https://paperswithcode.com/method/bert) inference. It inserts extra classification layers (which are referred to as off-ramps) between each [transformer](https://paperswithcode.com/method/transformer) layer of BERT. All transformer layers and off-ramps are jointly fine-tuned on a given downstream dataset. At inference time, after a sample goes through a transformer layer, it is passed to the following off-ramp. If the off-ramp is confident of the prediction, the result is returned; otherwise, the sample is sent to the next transformer layer.","description_state":"present","introduced_year":null,"introduced_by":{"title":"DeeBERT: Dynamic Early Exiting for Accelerating BERT Inference","paper":"/paper/deebert-dynamic-early-exiting-for","first_author":"Ji Xin","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/deebert-dynamic-early-exiting-for"},"source":{"url":"https://arxiv.org/abs/2004.12993v1","title":"DeeBERT: Dynamic Early Exiting for Accelerating BERT Inference","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Autoencoding Transformers","url":"/methods/category/autoencoding-transformers","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":"/paper/splitee-early-exit-in-deep-neural-networks","title":"SplitEE: Early Exit in Deep Neural Networks with Split Computing","date":"2023-09-17","arxiv_id":"2309.09195","n_code_links":1,"syntology":null},{"paper":"/paper/romebert-robust-training-of-multi-exit-bert","title":"RomeBERT: Robust Training of Multi-Exit BERT","date":"2021-01-24","arxiv_id":"2101.09755","n_code_links":1,"syntology":null},{"paper":"/paper/deebert-dynamic-early-exiting-for","title":"DeeBERT: Dynamic Early Exiting for Accelerating BERT Inference","date":"2020-04-27","arxiv_id":"2004.12993","n_code_links":3,"syntology":{"ran":5,"of":6,"unverified":1,"pointer_only":5}}],"papers_shown":3,"tasks":[{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":1},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":1},{"task":"/task/paraphrase-identification","name":"Paraphrase Identification","papers":1}],"tasks_shown":3,"n_tasks":3,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":1},{"year":"2023","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/deebert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}