{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/patient-knowledge-distillation-for-bert-model","title":"Patient Knowledge Distillation for BERT Model Compression","arxiv_id":"1908.09355","date":"2019-08-25","proceeding":"IJCNLP 2019 11","authors":["Siqi Sun","Yu Cheng","Zhe Gan","Jingjing Liu"],"abstract":"Pre-trained language models such as BERT have proven to be highly effective for natural language processing (NLP) tasks. However, the high demand for computing resources in training such models hinders their application in practice. In order to alleviate this resource hunger in large-scale model training, we propose a Patient Knowledge Distillation approach to compress an original large model (teacher) into an equally-effective lightweight shallow network (student). Different from previous knowledge distillation methods, which only use the output from the last layer of the teacher network for distillation, our student model patiently learns from multiple intermediate layers of the teacher model for incremental knowledge extraction, following two strategies: ($i$) PKD-Last: learning from the last $k$ layers; and ($ii$) PKD-Skip: learning from every $k$ layers. These two patient distillation schemes enable the exploitation of rich information in the teacher's hidden layers, and encourage the student model to patiently learn from and imitate the teacher through a multi-layer distillation process. Empirically, this translates into improved results on multiple NLP tasks with significant gain in training efficiency, without sacrificing model accuracy.","url_abs":"https://arxiv.org/abs/1908.09355v1","url_pdf":"https://arxiv.org/pdf/1908.09355v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"patient-knowledge-distillation-for-bert-model","repo_url":"https://github.com/intersun/PKD-for-BERT-Model-Compression","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"patient-knowledge-distillation-for-bert-model","repo_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"patient-knowledge-distillation-for-bert-model","repo_url":"https://github.com/eunanomist/PKD_BERT","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"patient-knowledge-distillation-for-bert-model","repo_url":"https://github.com/MindSpore-scientific/code-11/tree/main/Patient2Vec-A-Personalized-Interpretable","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"mindspore","reach":null},{"paper_slug":"patient-knowledge-distillation-for-bert-model","repo_url":"https://github.com/MindSpore-scientific/code-13/tree/main/Patient2Vec-A-Personalized-Interpretable","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"mindspore","reach":null}],"tasks":[{"task_slug":"knowledge-distillation","task_name":"Knowledge Distillation"},{"task_slug":"model-compression","task_name":"Model Compression"},{"task_slug":"model","task_name":"model"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bert","method_name":"BERT"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"knowledge-distillation","method_name":"Knowledge Distillation"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"linear-warmup-with-linear-decay","method_name":"Linear Warmup With Linear Decay"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"weight-decay","method_name":"Weight Decay"},{"method_slug":"wordpiece","method_name":"WordPiece"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/1908.09355","atlas_url":"https://app.syntology.ai/?focus=1908.09355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.09355"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/intersun/PKD-for-BERT-Model-Compression","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/eunanomist/PKD_BERT","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/MindSpore-scientific/code-11/tree/main/Patient2Vec-A-Personalized-Interpretable","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/MindSpore-scientific/code-13/tree/main/Patient2Vec-A-Personalized-Interpretable","reach":null}],"summary":{"ran_draft_wrong":5,"ran_fixture":1,"ran_violates":1,"ran":8,"ran_honours":2,"unverified":10},"by_repo_kind":{"listed":{"samples":27,"ran":17,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":27,"samples":[{"code_sha256_prefix":"0f786c407fb1ee4c","entry":"swish","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/BERT/pytorch_pretrained_bert/modeling.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":2,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"0f786c407fb1ee4c"}},{"code_sha256_prefix":"cac113ca87b9d9f3","entry":"acc_and_f1","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/nli_data_processing.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/nli_data_processing.py","link_basis":"harvester_set","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"cac113ca87b9d9f3"}},{"code_sha256_prefix":"b1f9cebf0d177140","entry":"boolean_string","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/argument_parser.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/argument_parser.py","link_basis":"harvester_set","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"b1f9cebf0d177140"}},{"code_sha256_prefix":"35ef111fc00c21b4","entry":"build_tf_to_pytorch_map","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling_transfo_xl.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/modeling_transfo_xl.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"35ef111fc00c21b4"}},{"code_sha256_prefix":"379ade5858acdbfa","entry":"convert_examples_to_features","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/nli_data_processing.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/nli_data_processing.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"379ade5858acdbfa"}},{"code_sha256_prefix":"b30ee2907e5cd3a6","entry":"convert_examples_to_features","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/race_data_processing.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/race_data_processing.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"b30ee2907e5cd3a6"}},{"code_sha256_prefix":"3ea89c558af6c4b4","entry":"count_parameters","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/utils.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/utils.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"3ea89c558af6c4b4"}},{"code_sha256_prefix":"5ff7ab3a0c0d6da0","entry":"distillation_loss","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/KD_loss.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/KD_loss.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"5ff7ab3a0c0d6da0"}},{"code_sha256_prefix":"8d23fbe2b99b840b","entry":"gelu","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling_gpt2.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/BERT/pytorch_pretrained_bert/modeling_gpt2.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"8d23fbe2b99b840b"}},{"code_sha256_prefix":"fdc64f4c72036ae4","entry":"gelu","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/BERT/pytorch_pretrained_bert/modeling.py","link_basis":"harvester_set","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"fdc64f4c72036ae4"}},{"code_sha256_prefix":"615f129ea793785e","entry":"is_folder_empty","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/argument_parser.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/argument_parser.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"615f129ea793785e"}},{"code_sha256_prefix":"9e42bf8972501658","entry":"load_model","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/utils.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/utils.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"9e42bf8972501658"}},{"code_sha256_prefix":"16afb11eb02e8e88","entry":"patience_loss","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/KD_loss.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/KD_loss.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"16afb11eb02e8e88"}},{"code_sha256_prefix":"28d61d48898d30c4","entry":"read_mrc_examples","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/race_data_processing.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/race_data_processing.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"28d61d48898d30c4"}},{"code_sha256_prefix":"93228a3eb5c4d179","entry":"sample_logits","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling_transfo_xl_utilities.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/modeling_transfo_xl_utilities.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"93228a3eb5c4d179"}},{"code_sha256_prefix":"5eff22fa0a651276","entry":"url_to_filename","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/file_utils.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/file_utils.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"5eff22fa0a651276"}},{"code_sha256_prefix":"e0960afdb0d64aa7","entry":"warmup_linear","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/optimization.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/optimization.py","link_basis":"harvester_set","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"e0960afdb0d64aa7"}},{"code_sha256_prefix":"df9f10631fa5d322","entry":"cached_path","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/file_utils.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/file_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"df9f10631fa5d322"}},{"code_sha256_prefix":"6db16fe8f67e56b6","entry":"filename_to_url","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/file_utils.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/file_utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"6db16fe8f67e56b6"}},{"code_sha256_prefix":"01803af97cd5feef","entry":"fill_tensor","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/utils.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"01803af97cd5feef"}},{"code_sha256_prefix":"b1a7180868dd8d6a","entry":"load_tf_weights_in_bert","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/BERT/pytorch_pretrained_bert/modeling.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"b1a7180868dd8d6a"}},{"code_sha256_prefix":"a62604f618a8cc82","entry":"load_tf_weights_in_gpt2","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling_gpt2.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/BERT/pytorch_pretrained_bert/modeling_gpt2.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"a62604f618a8cc82"}},{"code_sha256_prefix":"46f305675277e757","entry":"load_tf_weights_in_openai_gpt","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling_openai.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/modeling_openai.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"46f305675277e757"}},{"code_sha256_prefix":"1ebc2bf2c3cddc4d","entry":"load_tf_weights_in_transfo_xl","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/modeling_transfo_xl.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/modeling_transfo_xl.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"1ebc2bf2c3cddc4d"}},{"code_sha256_prefix":"3c241ecfe3749a6d","entry":"simple_accuracy","repo":"Daniel-H-99/Patient-Knowledge-Distillation","repo_kind":"listed","path":"src/nli_data_processing.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/src/nli_data_processing.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"3c241ecfe3749a6d"}},{"code_sha256_prefix":"59e4e730a9dc5dc6","entry":"warmup_constant","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/optimization.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/optimization.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"59e4e730a9dc5dc6"}},{"code_sha256_prefix":"007a8d2955fb555d","entry":"warmup_cosine","repo":"eunanomist/PKD_BERT","repo_kind":"listed","path":"BERT/pytorch_pretrained_bert/optimization.py","file_url":"https://github.com/eunanomist/PKD_BERT/blob/HEAD/BERT/pytorch_pretrained_bert/optimization.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"mcp_get_code":{"code_sha256":"007a8d2955fb555d"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}