{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/cup-curriculum-curriculum-learning-on-model","title":"Cup Curriculum: Curriculum Learning on Model Capacity","arxiv_id":"2311.03956","date":"2023-11-07","proceeding":null,"authors":["Luca Scharr","Vanessa Toborek"],"abstract":"Curriculum learning (CL) aims to increase the performance of a learner on a given task by applying a specialized learning strategy. This strategy focuses on either the dataset, the task, or the model. There is little to no work analysing the possibilities to apply CL on the model capacity in natural language processing. To close this gap, we propose the cup curriculum. In a first phase of training we use a variation of iterative magnitude pruning to reduce model capacity. These weights are reintroduced in a second phase, resulting in the model capacity to show a cup-shaped curve over the training iterations. We empirically evaluate different strategies of the cup curriculum and show that it outperforms early stopping reliably while exhibiting a high resilience to overfitting.","url_abs":"https://arxiv.org/abs/2311.03956v1","url_pdf":"https://arxiv.org/pdf/2311.03956v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"cup-curriculum-curriculum-learning-on-model","repo_url":"https://github.com/luca-scharr/cupcurriculum","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"model","task_name":"model"}],"methods":[{"method_slug":"early-stopping","method_name":"Early Stopping"},{"method_slug":"pruning","method_name":"Pruning"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2311.03956","atlas_url":"https://app.syntology.ai/?focus=2311.03956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03956"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/luca-scharr/cupcurriculum","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":3},"by_repo_kind":{"official":{"samples":3,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"c7b84365cf3df5a0","entry":"generate_square_subsequent_mask","repo":"luca-scharr/cupcurriculum","repo_kind":"official","path":"transformer_modell.py","file_url":"https://github.com/luca-scharr/cupcurriculum/blob/HEAD/transformer_modell.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c7b84365cf3df5a0"}},{"code_sha256_prefix":"410038cda15750c9","entry":"hodges_lehmann_estimator","repo":"luca-scharr/cupcurriculum","repo_kind":"official","path":"plot_generator.py","file_url":"https://github.com/luca-scharr/cupcurriculum/blob/HEAD/plot_generator.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"410038cda15750c9"}},{"code_sha256_prefix":"6aeee88cf4630d23","entry":"print_nonzeros","repo":"luca-scharr/cupcurriculum","repo_kind":"official","path":"utils.py","file_url":"https://github.com/luca-scharr/cupcurriculum/blob/HEAD/utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"6aeee88cf4630d23"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}