{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/bert-of-theseus-compressing-bert-by","title":"BERT-of-Theseus: Compressing BERT by Progressive Module Replacing","arxiv_id":"2002.02925","date":"2020-02-07","proceeding":"EMNLP 2020 11","authors":["Canwen Xu","Wangchunshu Zhou","Tao Ge","Furu Wei","Ming Zhou"],"abstract":"In this paper, we propose a novel model compression approach to effectively compress BERT by progressive module replacing. Our approach first divides the original BERT into several modules and builds their compact substitutes. Then, we randomly replace the original modules with their substitutes to train the compact modules to mimic the behavior of the original modules. We progressively increase the probability of replacement through the training. In this way, our approach brings a deeper level of interaction between the original and compact models. Compared to the previous knowledge distillation approaches for BERT compression, our approach does not introduce any additional loss function. Our approach outperforms existing knowledge distillation approaches on GLUE benchmark, showing a new perspective of model compression.","url_abs":"https://arxiv.org/abs/2002.02925v4","url_pdf":"https://arxiv.org/pdf/2002.02925v4.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"bert-of-theseus-compressing-bert-by","repo_url":"https://github.com/JetRunner/BERT-of-Theseus","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"bert-of-theseus-compressing-bert-by","repo_url":"https://github.com/ambroggi/pruning-project-for-deep-neural-networks","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"knowledge-distillation","task_name":"Knowledge Distillation"},{"task_slug":"model-compression","task_name":"Model Compression"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bert","method_name":"BERT"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"knowledge-distillation","method_name":"Knowledge Distillation"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"linear-warmup-with-linear-decay","method_name":"Linear Warmup With Linear Decay"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"weight-decay","method_name":"Weight Decay"},{"method_slug":"wordpiece","method_name":"WordPiece"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2002.02925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02925"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ambroggi/pruning-project-for-deep-neural-networks","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/JetRunner/BERT-of-Theseus","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"ran":1,"unverified":4},"by_repo_kind":{"official":{"samples":4,"ran":0,"repositories":1},"listed":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"62cf76535eb8767c","entry":"Theseus_Replacement","repo":"ambroggi/pruning-project-for-deep-neural-networks","repo_kind":"listed","path":"src/Imported_Code/BERT_Theseus_From_Paper.py","file_url":"https://github.com/ambroggi/pruning-project-for-deep-neural-networks/blob/HEAD/src/Imported_Code/BERT_Theseus_From_Paper.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"62cf76535eb8767c"}},{"code_sha256_prefix":"f19fe05d830bc91a","entry":"evaluate","repo":"JetRunner/BERT-of-Theseus","repo_kind":"official","path":"run_glue.py","file_url":"https://github.com/JetRunner/BERT-of-Theseus/blob/HEAD/run_glue.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f19fe05d830bc91a"}},{"code_sha256_prefix":"e39ee757d6852410","entry":"evaluate","repo":"JetRunner/BERT-of-Theseus","repo_kind":"official","path":"glue_script/run_prediction.py","file_url":"https://github.com/JetRunner/BERT-of-Theseus/blob/HEAD/glue_script/run_prediction.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"e39ee757d6852410"}},{"code_sha256_prefix":"794b556e7b37a67d","entry":"load_and_cache_examples","repo":"JetRunner/BERT-of-Theseus","repo_kind":"official","path":"run_glue.py","file_url":"https://github.com/JetRunner/BERT-of-Theseus/blob/HEAD/run_glue.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"794b556e7b37a67d"}},{"code_sha256_prefix":"afc76339aaf5823b","entry":"load_and_cache_examples","repo":"JetRunner/BERT-of-Theseus","repo_kind":"official","path":"glue_script/run_prediction.py","file_url":"https://github.com/JetRunner/BERT-of-Theseus/blob/HEAD/glue_script/run_prediction.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"afc76339aaf5823b"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}