{"url":"/method/early-exiting","slug":"early-exiting","name":"Early exiting","full_name":"Early exiting using confidence measures","full_name_withheld":false,"description_markdown":"Exit whenever the model is confident enough allowing early exiting from hidden layers","description_state":"present","introduced_year":null,"introduced_by":{"title":"BranchyNet: Fast Inference via Early Exiting from Deep Neural Networks","paper":"/paper/branchynet-fast-inference-via-early-exiting","first_author":"Surat Teerapittayanon","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/branchynet-fast-inference-via-early-exiting"},"source":{"url":"http://arxiv.org/abs/1709.01686v1","title":"BranchyNet: Fast Inference via Early Exiting from Deep Neural Networks","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Loss Functions","url":"/methods/category/loss-functions","pwc_aliases":[]}],"n_papers_tagged":66,"archive_num_papers":66,"papers_newest_first":[{"paper":null,"title":"Early Stopping Tabular In-Context Learning","date":"2025-06-26","arxiv_id":"2506.21387","n_code_links":0,"syntology":null},{"paper":null,"title":"Languages in Multilingual Speech Foundation Models Align Both Phonetically and Semantically","date":"2025-05-26","arxiv_id":"2505.19606","n_code_links":0,"syntology":null},{"paper":null,"title":"System-1.5 Reasoning: Traversal in Language and Latent Spaces with Dynamic Shortcuts","date":"2025-05-25","arxiv_id":"2505.18962","n_code_links":0,"syntology":null},{"paper":null,"title":"Performance Control in Early Exiting to Deploy Large Models at the Same Cost of Smaller Ones","date":"2024-12-26","arxiv_id":"2412.19325","n_code_links":0,"syntology":null},{"paper":"/paper/cosee-consistency-oriented-signal-based-early","title":"COSEE: Consistency-Oriented Signal-Based Early Exiting via Calibrated Sample Weighting Mechanism","date":"2024-12-17","arxiv_id":"2412.13236","n_code_links":1,"syntology":null},{"paper":"/paper/a-stitch-in-time-saves-nine-small-vlm-is-a","title":"A Stitch in Time Saves Nine: Small VLM is a Precise Guidance for Accelerating Large VLMs","date":"2024-12-04","arxiv_id":"2412.03324","n_code_links":1,"syntology":{"ran":4,"of":5,"unverified":1,"pointer_only":5}},{"paper":null,"title":"Energy-Aware Dynamic Neural Inference","date":"2024-11-04","arxiv_id":"2411.02471","n_code_links":0,"syntology":null},{"paper":null,"title":"BERTer: The Efficient One","date":"2024-07-19","arxiv_id":"2407.14039","n_code_links":0,"syntology":null},{"paper":null,"title":"Tracing Representation Progression: Analyzing and Enhancing Layer-Wise Similarity","date":"2024-06-20","arxiv_id":"2406.14479","n_code_links":0,"syntology":null},{"paper":null,"title":"Predicting Probabilities of Error to Combine Quantization and Early Exiting: QuEE","date":"2024-06-20","arxiv_id":"2406.14404","n_code_links":0,"syntology":null},{"paper":null,"title":"RAEE: A Robust Retrieval-Augmented Early Exiting Framework for Efficient Inference","date":"2024-05-24","arxiv_id":"2405.15198","n_code_links":0,"syntology":null},{"paper":"/paper/a-comprehensive-survey-of-accelerated","title":"A Comprehensive Survey of Accelerated Generation Techniques in Large Language Models","date":"2024-05-15","arxiv_id":"2405.13019","n_code_links":1,"syntology":null},{"paper":null,"title":"Computational Efficient Width-Wise Early Exiting in Wireless Communication Systems","date":"2024-05-06","arxiv_id":"2405.03222","n_code_links":0,"syntology":null},{"paper":"/paper/kangaroo-lossless-self-speculative-decoding","title":"Kangaroo: Lossless Self-Speculative Decoding via Double Early Exiting","date":"2024-04-29","arxiv_id":"2404.18911","n_code_links":1,"syntology":{"ran":4,"of":6,"unverified":2,"pointer_only":6}},{"paper":"/paper/layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","n_code_links":1,"syntology":{"ran":4,"of":5,"unverified":1,"pointer_only":5}},{"paper":"/paper/efficient-transformer-encoders-for","title":"Efficient Transformer Encoders for Mask2Former-style models","date":"2024-04-23","arxiv_id":"2404.15244","n_code_links":1,"syntology":null},{"paper":"/paper/tiny-models-are-the-computational-saver-for","title":"Tiny Models are the Computational Saver for Large Models","date":"2024-03-26","arxiv_id":"2403.17726","n_code_links":1,"syntology":null},{"paper":null,"title":"Understanding the Training Speedup from Sampling with Approximate Losses","date":"2024-02-10","arxiv_id":"2402.07052","n_code_links":0,"syntology":null},{"paper":null,"title":"EERO: Early Exit with Reject Option for Efficient Classification with limited budget","date":"2024-02-06","arxiv_id":"2402.03779","n_code_links":0,"syntology":null},{"paper":null,"title":"DE$^3$-BERT: Distance-Enhanced Early Exiting for BERT based on Prototypical Networks","date":"2024-02-03","arxiv_id":"2402.05948","n_code_links":0,"syntology":null},{"paper":"/paper/consistentee-a-consistent-and-hardness-guided","title":"ConsistentEE: A Consistent and Hardness-Guided Early Exiting Method for Accelerating Language Models Inference","date":"2023-12-19","arxiv_id":"2312.11882","n_code_links":1,"syntology":null},{"paper":"/paper/ee-llm-large-scale-training-and-inference-of","title":"EE-LLM: Large-Scale Training and Inference of Early-Exit Large Language Models with 3D Parallelism","date":"2023-12-08","arxiv_id":"2312.04916","n_code_links":1,"syntology":{"ran":3,"of":5,"unverified":2,"pointer_only":5}},{"paper":null,"title":"Adaptive Early Exiting for Collaborative Inference over Noisy Wireless Channels","date":"2023-11-29","arxiv_id":"2311.18098","n_code_links":0,"syntology":null},{"paper":null,"title":"Accelerating LLaMA Inference by Enabling Intermediate Layer Decoding via Instruction Tuning with LITE","date":"2023-10-28","arxiv_id":"2310.18581","n_code_links":0,"syntology":null},{"paper":null,"title":"AdaDiff: Accelerating Diffusion Models through Step-Wise Adaptive Computation","date":"2023-09-29","arxiv_id":"2309.17074","n_code_links":0,"syntology":null},{"paper":null,"title":"Dynamic Early Exiting Predictive Coding Neural Networks","date":"2023-09-05","arxiv_id":"2309.02022","n_code_links":0,"syntology":null},{"paper":null,"title":"Using Early Exits for Fast Inference in Automatic Modulation Classification","date":"2023-08-22","arxiv_id":"2308.11100","n_code_links":0,"syntology":null},{"paper":"/paper/lgvit-dynamic-early-exiting-for-accelerating","title":"LGViT: Dynamic Early Exiting for Accelerating Vision Transformer","date":"2023-08-01","arxiv_id":"2308.00255","n_code_links":1,"syntology":null},{"paper":null,"title":"SkipDecode: Autoregressive Skip Decoding with Batching and Caching for Efficient LLM Inference","date":"2023-07-05","arxiv_id":"2307.02628","n_code_links":0,"syntology":null},{"paper":null,"title":"CSI-Based Efficient Self-Quarantine Monitoring System Using Branchy Convolution Neural Network","date":"2023-05-24","arxiv_id":"2306.01756","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":5},{"task":"/task/image-classification","name":"Image Classification","papers":4},{"task":"/task/language-modeling","name":"Language Modeling","papers":4},{"task":"/task/text-generation","name":"Text Generation","papers":4},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":3},{"task":"/task/decoder","name":"Decoder","papers":3},{"task":"/task/edge-computing","name":"Edge-computing","papers":3},{"task":"/task/model-compression","name":"Model Compression","papers":3},{"task":"/task/retrieval","name":"Retrieval","papers":3},{"task":"/task/text-classification","name":"Text Classification","papers":3},{"task":"/task/image-classification","name":"image-classification","papers":3},{"task":"/task/text-classification-1","name":"text-classification","papers":3},{"task":"/task/collaborative-inference","name":"Collaborative Inference","papers":2},{"task":"/task/gsm8k","name":"GSM8K","papers":2},{"task":"/task/model-selection","name":"Model Selection","papers":2},{"task":"/task/named-entity-recognition-1","name":"Named Entity Recognition","papers":2},{"task":"/task/named-entity-recognition-ner","name":"Named Entity Recognition (NER)","papers":2},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":2},{"task":"/task/object-detection","name":"Object Detection","papers":2},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":2}],"tasks_shown":20,"n_tasks":86,"usage_by_year":[{"year":"2017","papers":1},{"year":"2019","papers":3},{"year":"2020","papers":5},{"year":"2021","papers":11},{"year":"2022","papers":11},{"year":"2023","papers":15},{"year":"2024","papers":17},{"year":"2025","papers":3}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/early-exiting"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}