{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/preserving-generalization-of-language-models","title":"Preserving Generalization of Language models in Few-shot Continual Relation Extraction","arxiv_id":"2410.00334","date":"2024-10-01","proceeding":null,"authors":["Quyen Tran","Nguyen Xuan Thanh","Nguyen Hoang Anh","Nam Le Hai","Trung Le","Linh Van Ngo","Thien Huu Nguyen"],"abstract":"Few-shot Continual Relations Extraction (FCRE) is an emerging and dynamic area of study where models can sequentially integrate knowledge from new relations with limited labeled data while circumventing catastrophic forgetting and preserving prior knowledge from pre-trained backbones. In this work, we introduce a novel method that leverages often-discarded language model heads. By employing these components via a mutual information maximization strategy, our approach helps maintain prior knowledge from the pre-trained backbone and strategically aligns the primary classification head, thereby enhancing model performance. Furthermore, we explore the potential of Large Language Models (LLMs), renowned for their wealth of knowledge, in addressing FCRE challenges. Our comprehensive experimental results underscore the efficacy of the proposed method and offer valuable insights for future work.","url_abs":"https://arxiv.org/abs/2410.00334v1","url_pdf":"https://arxiv.org/pdf/2410.00334v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"preserving-generalization-of-language-models","repo_url":"https://github.com/thanhnx12/cre-via-mmi","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"jax","reach":{"status":"ok"}}],"tasks":[{"task_slug":"continual-relation-extraction","task_name":"Continual Relation Extraction"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":null,"task_name":"Relation"},{"task_slug":"relation-extraction","task_name":"Relation Extraction"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2410.00334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00334"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/thanhnx12/CRE-via-MMI","reach":{"status":"ok"}}],"summary":{"ran":4},"by_repo_kind":{"official":{"samples":4,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"37b5a9dbfddf1f9a","entry":"compute_jsd_loss","repo":"thanhnx12/CRE-via-MMI","repo_kind":"official","path":"SCKD/main-llm-mmi.py","file_url":"https://github.com/thanhnx12/CRE-via-MMI/blob/HEAD/SCKD/main-llm-mmi.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"37b5a9dbfddf1f9a"}},{"code_sha256_prefix":"02c3e2a3a558a690","entry":"construct_hard_triplets","repo":"thanhnx12/CRE-via-MMI","repo_kind":"official","path":"SCKD/main-llm-mmi.py","file_url":"https://github.com/thanhnx12/CRE-via-MMI/blob/HEAD/SCKD/main-llm-mmi.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"02c3e2a3a558a690"}},{"code_sha256_prefix":"4c858d28c511c74d","entry":"contrastive_loss","repo":"thanhnx12/CRE-via-MMI","repo_kind":"official","path":"SCKD/main-llm-mmi.py","file_url":"https://github.com/thanhnx12/CRE-via-MMI/blob/HEAD/SCKD/main-llm-mmi.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"4c858d28c511c74d"}},{"code_sha256_prefix":"28bd61afce7c0a27","entry":"get_data_loader_BERT","repo":"thanhnx12/CRE-via-MMI","repo_kind":"official","path":"CPL/data_loader.py","file_url":"https://github.com/thanhnx12/CRE-via-MMI/blob/HEAD/CPL/data_loader.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"28bd61afce7c0a27"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}