{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/abrupt-learning-in-transformers-a-case-study","title":"Abrupt Learning in Transformers: A Case Study on Matrix Completion","arxiv_id":"2410.22244","date":"2024-10-29","proceeding":null,"authors":["Pulkit Gopalani","Ekdeep Singh Lubana","Wei Hu"],"abstract":"Recent analysis on the training dynamics of Transformers has unveiled an interesting characteristic: the training loss plateaus for a significant number of training steps, and then suddenly (and sharply) drops to near--optimal values. To understand this phenomenon in depth, we formulate the low-rank matrix completion problem as a masked language modeling (MLM) task, and show that it is possible to train a BERT model to solve this task to low error. Furthermore, the loss curve shows a plateau early in training followed by a sudden drop to near-optimal values, despite no changes in the training procedure or hyper-parameters. To gain interpretability insights into this sudden drop, we examine the model's predictions, attention heads, and hidden states before and after this transition. Concretely, we observe that (a) the model transitions from simply copying the masked input to accurately predicting the masked entries; (b) the attention heads transition to interpretable patterns relevant to the task; and (c) the embeddings and hidden states encode information relevant to the problem. We also analyze the training dynamics of individual model components to understand the sudden drop in loss.","url_abs":"https://arxiv.org/abs/2410.22244v1","url_pdf":"https://arxiv.org/pdf/2410.22244v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"low-rank-matrix-completion","task_name":"Low-Rank Matrix Completion"},{"task_slug":"masked-language-modeling","task_name":"Masked Language Modeling"},{"task_slug":"matrix-completion","task_name":"Matrix Completion"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bert","method_name":"BERT"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"linear-warmup-with-linear-decay","method_name":"Linear Warmup With Linear Decay"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"weight-decay","method_name":"Weight Decay"},{"method_slug":"wordpiece","method_name":"WordPiece"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2410.22244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22244"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/pulkitgopalani/tf-matcomp","reach":{"status":"ok"}}],"summary":{"unverified":4},"by_repo_kind":{"found_in_text":{"samples":4,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"d1ff93c599a81f10","entry":"nuc_norm_alt","repo":"pulkitgopalani/tf-matcomp","repo_kind":"found_in_text","path":"src/cvx_nuc_norm.py","file_url":"https://github.com/pulkitgopalani/tf-matcomp/blob/HEAD/src/cvx_nuc_norm.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"d1ff93c599a81f10"}},{"code_sha256_prefix":"5f8df68b6eb2411c","entry":"nuc_norm_solver","repo":"pulkitgopalani/tf-matcomp","repo_kind":"found_in_text","path":"src/cvx_nuc_norm.py","file_url":"https://github.com/pulkitgopalani/tf-matcomp/blob/HEAD/src/cvx_nuc_norm.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5f8df68b6eb2411c"}},{"code_sha256_prefix":"87164d0476074ab4","entry":"rank_approx","repo":"pulkitgopalani/tf-matcomp","repo_kind":"found_in_text","path":"src/copy_check.py","file_url":"https://github.com/pulkitgopalani/tf-matcomp/blob/HEAD/src/copy_check.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"87164d0476074ab4"}},{"code_sha256_prefix":"594821afeae903a9","entry":"reg_mse_solver","repo":"pulkitgopalani/tf-matcomp","repo_kind":"found_in_text","path":"src/cvx_nuc_norm.py","file_url":"https://github.com/pulkitgopalani/tf-matcomp/blob/HEAD/src/cvx_nuc_norm.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"594821afeae903a9"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}