{"url":"/task/continual-pretraining","name":"Continual Pretraining","slug":"continual-pretraining","description_markdown":null,"categories":[{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":70,"papers_with_code":34,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/continual-pretraining-on-acl-arc","slug":"continual-pretraining-on-acl-arc","dataset":"ACL-ARC","dataset_url":"/dataset/acl-arc-1","rows_in_archive":1,"metrics":["F1 (macro)"],"first_row_in_archive_order":{"model":"DAS","paper_title":"Continual Pre-training of Language Models","paper_url":"/paper/continual-learning-of-language-models","paper_date":"2023-02-07","arxiv_id":"2302.03241","code_links":[{"title":"zixuanke/pycontinual","url":"https://github.com/zixuanke/pycontinual"},{"title":"UIC-Liu-Lab/ContinualLM","url":"https://github.com/UIC-Liu-Lab/ContinualLM"}],"syntology":null}},{"leaderboard":"/sota/continual-pretraining-on-ag-news","slug":"continual-pretraining-on-ag-news","dataset":"AG News","dataset_url":"/dataset/ag-news","rows_in_archive":1,"metrics":["F1 - macro"],"first_row_in_archive_order":{"model":"CPT","paper_title":"Continual Training of Language Models for Few-Shot Learning","paper_url":"/paper/continual-training-of-language-models-for-few","paper_date":"2022-10-11","arxiv_id":"2210.05549","code_links":[{"title":"zixuanke/pycontinual","url":"https://github.com/zixuanke/pycontinual"},{"title":"UIC-Liu-Lab/ContinualLM","url":"https://github.com/UIC-Liu-Lab/ContinualLM"},{"title":"uic-liu-lab/cpt","url":"https://github.com/uic-liu-lab/cpt"}],"syntology":null}},{"leaderboard":"/sota/continual-pretraining-on-scierc","slug":"continual-pretraining-on-scierc","dataset":"SciERC","dataset_url":"/dataset/scierc","rows_in_archive":1,"metrics":["F1 (macro)"],"first_row_in_archive_order":{"model":"DAS","paper_title":"Continual Pre-training of Language Models","paper_url":"/paper/continual-learning-of-language-models","paper_date":"2023-02-07","arxiv_id":"2302.03241","code_links":[{"title":"zixuanke/pycontinual","url":"https://github.com/zixuanke/pycontinual"},{"title":"UIC-Liu-Lab/ContinualLM","url":"https://github.com/UIC-Liu-Lab/ContinualLM"}],"syntology":null}}],"datasets":[{"url":"/dataset/ag-news","name":"AG News","full_name":"AG’s News Corpus","num_papers_in_archive":969},{"url":"/dataset/scierc","name":"SciERC","full_name":"","num_papers_in_archive":134},{"url":"/dataset/acl-arc-1","name":"ACL ARC","full_name":"","num_papers_in_archive":14}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":34,"tagged_in_all":70,"items":[{"url":"/paper/rho-1-not-all-tokens-are-what-you-need","title":"Rho-1: Not All Tokens Are What You Need","date":"2024-04-11","arxiv_id":"2404.07965","repositories_listed":3,"syntology":null},{"url":"/paper/continual-training-of-language-models-for-few","title":"Continual Training of Language Models for Few-Shot Learning","date":"2022-10-11","arxiv_id":"2210.05549","repositories_listed":3,"syntology":null},{"url":"/paper/automathtext-autonomous-data-selection-with","title":"Autonomous Data Selection with Zero-shot Generative Classifiers for Mathematical Texts","date":"2024-02-12","arxiv_id":"2402.07625","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/effective-long-context-scaling-of-foundation","title":"Effective Long-Context Scaling of Foundation Models","date":"2023-09-27","arxiv_id":"2309.16039","repositories_listed":2,"syntology":null},{"url":"/paper/gfm-building-geospatial-foundation-models-via","title":"Towards Geospatial Foundation Models via Continual Pretraining","date":"2023-02-09","arxiv_id":"2302.04476","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/continual-learning-of-language-models","title":"Continual Pre-training of Language Models","date":"2023-02-07","arxiv_id":"2302.03241","repositories_listed":2,"syntology":null},{"url":"/paper/deer-a-data-efficient-language-model-for","title":"ECONET: Effective Continual Pretraining of Language Models for Event Temporal Reasoning","date":"2020-12-30","arxiv_id":"2012.15283","repositories_listed":2,"syntology":null},{"url":"/paper/domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","arxiv_id":"2007.15779","repositories_listed":2,"syntology":null},{"url":"/paper/simulating-training-data-leakage-in-multiple","title":"Simulating Training Data Leakage in Multiple-Choice Benchmarks for LLM Evaluation","date":"2025-05-30","arxiv_id":"2505.24263","repositories_listed":1,"syntology":null},{"url":"/paper/a-japanese-language-model-and-three-new","title":"A Japanese Language Model and Three New Evaluation Benchmarks for Pharmaceutical NLP","date":"2025-05-22","arxiv_id":"2505.16661","repositories_listed":1,"syntology":null},{"url":"/paper/tic-lm-a-web-scale-benchmark-for-time","title":"TiC-LM: A Web-Scale Benchmark for Time-Continual LLM Pretraining","date":"2025-04-02","arxiv_id":"2504.02107","repositories_listed":1,"syntology":null},{"url":"/paper/robust-data-watermarking-in-language-models","title":"Robust Data Watermarking in Language Models by Injecting Fictitious Knowledge","date":"2025-03-06","arxiv_id":"2503.04036","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-domain-adaptive-post-training","title":"Demystifying Domain-adaptive Post-training for Financial LLMs","date":"2025-01-09","arxiv_id":"2501.04961","repositories_listed":1,"syntology":{"n":13,"n_ran":4,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/nyayaanumana-inlegalllama-the-largest-indian","title":"NyayaAnumana & INLegalLlama: The Largest Indian Legal Judgment Prediction Dataset and Specialized Language Model for Enhanced Decision Analysis","date":"2024-12-11","arxiv_id":"2412.08385","repositories_listed":1,"syntology":null},{"url":"/paper/alchemy-amplifying-theorem-proving-capability","title":"Alchemy: Amplifying Theorem-Proving Capability through Symbolic Mutation","date":"2024-10-21","arxiv_id":"2410.15748","repositories_listed":1,"syntology":null},{"url":"/paper/langsamp-language-script-aware-multilingual","title":"LangSAMP: Language-Script Aware Multilingual Pretraining","date":"2024-09-26","arxiv_id":"2409.18199","repositories_listed":1,"syntology":null},{"url":"/paper/a-practitioner-s-guide-to-continual","title":"A Practitioner's Guide to Continual Multimodal Pretraining","date":"2024-08-26","arxiv_id":"2408.14471","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-granite-code-models-to-128k-context","title":"Scaling Granite Code Models to 128K Context","date":"2024-07-18","arxiv_id":"2407.13739","repositories_listed":1,"syntology":null},{"url":"/paper/towards-lifelong-learning-of-large-language","title":"Towards Lifelong Learning of Large Language Models: A Survey","date":"2024-06-10","arxiv_id":"2406.06391","repositories_listed":1,"syntology":null},{"url":"/paper/multi-label-guided-soft-contrastive-learning","title":"Multi-Label Guided Soft Contrastive Learning for Efficient Earth Observation Pretraining","date":"2024-05-30","arxiv_id":"2405.20462","repositories_listed":1,"syntology":null},{"url":"/paper/mora-high-rank-updating-for-parameter","title":"MoRA: High-Rank Updating for Parameter-Efficient Fine-Tuning","date":"2024-05-20","arxiv_id":"2405.12130","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/retrieval-head-mechanistically-explains-long","title":"Retrieval Head Mechanistically Explains Long-Context Factuality","date":"2024-04-24","arxiv_id":"2404.15574","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":7}},{"url":"/paper/yi-open-foundation-models-by-01-ai","title":"Yi: Open Foundation Models by 01.AI","date":"2024-03-07","arxiv_id":"2403.04652","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/data-engineering-for-scaling-language-models","title":"Data Engineering for Scaling Language Models to 128K Context","date":"2024-02-15","arxiv_id":"2402.10171","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/pecop-parameter-efficient-continual","title":"PECoP: Parameter Efficient Continual Pretraining for Action Quality Assessment","date":"2023-11-11","arxiv_id":"2311.07603","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/ctp-towards-vision-language-continual","title":"CTP: Towards Vision-Language Continual Pretraining via Compatible Momentum Contrast and Topology Preservation","date":"2023-08-14","arxiv_id":"2308.07146","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/ctp-towards-vision-language-continual-1","title":"CTP:Towards Vision-Language Continual Pretraining via Compatible Momentum Contrast and Topology Preservation","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cbeaf-adapting-enhanced-continual-pretraining","title":"AF Adapter: Continual Pretraining for Building Chinese Biomedical Language Model","date":"2022-11-21","arxiv_id":"2211.11363","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-domain-adaptation-for-sparse","title":"Unsupervised Domain Adaptation for Sparse Retrieval by Filling Vocabulary and Word Frequency Gaps","date":"2022-11-08","arxiv_id":"2211.03988","repositories_listed":1,"syntology":null},{"url":"/paper/continual-pre-training-mitigates-forgetting","title":"Continual Pre-Training Mitigates Forgetting in Language and Vision","date":"2022-05-19","arxiv_id":"2205.09357","repositories_listed":1,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}