{"url":"/method/inverse-square-root-schedule","slug":"inverse-square-root-schedule","name":"Inverse Square Root Schedule","full_name":"Inverse Square Root Schedule","full_name_withheld":false,"description_markdown":"**Inverse Square Root** is a learning rate schedule 1 / $\\sqrt{\\max\\left(n, k\\right)}$ where\r\n$n$ is the current training iteration and $k$ is the number of warm-up steps. This sets a constant learning rate for the first $k$ steps, then exponentially decays the learning rate until pre-training is over.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":null,"title":null,"url_on_a_paper_host":false},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Learning Rate Schedules","url":"/methods/category/learning-rate-schedules","pwc_aliases":[]}],"n_papers_tagged":702,"archive_num_papers":702,"papers_newest_first":[{"paper":null,"title":"Chat-Ghosting: A Comparative Study of Methods for Auto-Completion in Dialog Systems","date":"2025-07-08","arxiv_id":"2507.05940","n_code_links":0,"syntology":null},{"paper":null,"title":"I Know Which LLM Wrote Your Code Last Summer: LLM generated Code Stylometry for Authorship Attribution","date":"2025-06-18","arxiv_id":"2506.17323","n_code_links":0,"syntology":null},{"paper":null,"title":"Fretting-Transformer: Encoder-Decoder Model for MIDI to Tablature Transcription","date":"2025-06-17","arxiv_id":"2506.14223","n_code_links":0,"syntology":null},{"paper":null,"title":"A Comprehensive Study of Decoder-Only LLMs for Text-to-Image Generation","date":"2025-06-09","arxiv_id":"2506.08210","n_code_links":0,"syntology":null},{"paper":null,"title":"A Multi-Dataset Evaluation of Models for Automated Vulnerability Repair","date":"2025-06-05","arxiv_id":"2506.04987","n_code_links":0,"syntology":null},{"paper":null,"title":"Decom-Renorm-Merge: Model Merging on the Right Space Improves Multitasking","date":"2025-05-29","arxiv_id":"2505.23117","n_code_links":0,"syntology":null},{"paper":"/paper/shioenv-a-cli-behavior-capturing-environment","title":"ShIOEnv: A CLI Behavior-Capturing Environment Enabling Grammar-Guided Command Synthesis for Dataset Curation","date":"2025-05-23","arxiv_id":"2505.18374","n_code_links":1,"syntology":null},{"paper":"/paper/logicase-effective-test-case-generation-from","title":"LogiCase: Effective Test Case Generation from Logical Description in Competitive Programming","date":"2025-05-21","arxiv_id":"2505.15039","n_code_links":0,"syntology":{"ran":11,"of":19,"unverified":8,"pointer_only":19}},{"paper":"/paper/eeg-to-text-translation-a-model-for","title":"EEG-to-Text Translation: A Model for Deciphering Human Brain Activity","date":"2025-05-20","arxiv_id":"2505.13936","n_code_links":1,"syntology":null},{"paper":"/paper/masking-in-multi-hop-qa-an-analysis-of-how","title":"Masking in Multi-hop QA: An Analysis of How Language Models Perform with Context Permutation","date":"2025-05-16","arxiv_id":"2505.11754","n_code_links":1,"syntology":null},{"paper":null,"title":"Multilingual Machine Translation with Quantum Encoder Decoder Attention-based Convolutional Variational Circuits","date":"2025-05-14","arxiv_id":"2505.09407","n_code_links":0,"syntology":null},{"paper":null,"title":"Performance Evaluation of Large Language Models in Bangla Consumer Health Query Summarization","date":"2025-05-08","arxiv_id":"2505.05070","n_code_links":0,"syntology":null},{"paper":"/paper/gascade-grouped-summarization-of-adverse-drug","title":"GASCADE: Grouped Summarization of Adverse Drug Event for Enhanced Cancer Pharmacovigilance","date":"2025-05-07","arxiv_id":"2505.04284","n_code_links":1,"syntology":null},{"paper":null,"title":"A review of DNA restriction-free overlapping sequence cloning techniques for synthetic biology","date":"2025-05-06","arxiv_id":"2505.03681","n_code_links":0,"syntology":null},{"paper":null,"title":"JaccDiv: A Metric and Benchmark for Quantifying Diversity of Generated Marketing Text in the Music Industry","date":"2025-04-29","arxiv_id":"2504.20849","n_code_links":0,"syntology":null},{"paper":null,"title":"Large Language Models are Qualified Benchmark Builders: Rebuilding Pre-Training Datasets for Advancing Code Intelligence Tasks","date":"2025-04-28","arxiv_id":"2504.19444","n_code_links":0,"syntology":null},{"paper":"/paper/sigma-a-dataset-for-text-to-code-semantic","title":"Sigma: A dataset for text-to-code semantic parsing with statistical analysis","date":"2025-04-05","arxiv_id":"2504.04301","n_code_links":1,"syntology":null},{"paper":null,"title":"Advancing Sentiment Analysis in Tamil-English Code-Mixed Texts: Challenges and Transformer-Based Solutions","date":"2025-03-30","arxiv_id":"2503.23295","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Knowledge Graph Completion with Entity Neighborhood and Relation Context","date":"2025-03-29","arxiv_id":"2503.23205","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-down-text-encoders-of-text-to-image","title":"Scaling Down Text Encoders of Text-to-Image Diffusion Models","date":"2025-03-25","arxiv_id":"2503.19897","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-training-and-inference-scaling-laws","title":"Exploring Training and Inference Scaling Laws in Generative Retrieval","date":"2025-03-24","arxiv_id":"2503.18941","n_code_links":1,"syntology":null},{"paper":null,"title":"Predicting the Road Ahead: A Knowledge Graph based Foundation Model for Scene Understanding in Autonomous Driving","date":"2025-03-24","arxiv_id":"2503.18730","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Code LLM Training with Programmer Attention","date":"2025-03-19","arxiv_id":"2503.14936","n_code_links":0,"syntology":null},{"paper":null,"title":"DreamRenderer: Taming Multi-Instance Attribute Control in Large-Scale Text-to-Image Models","date":"2025-03-17","arxiv_id":"2503.12885","n_code_links":0,"syntology":null},{"paper":null,"title":"ARLED: Leveraging LED-based ARMAN Model for Abstractive Summarization of Persian Long Documents","date":"2025-03-13","arxiv_id":"2503.10233","n_code_links":0,"syntology":null},{"paper":null,"title":"A LongFormer-Based Framework for Accurate and Efficient Medical Text Summarization","date":"2025-03-10","arxiv_id":"2503.06888","n_code_links":0,"syntology":null},{"paper":"/paper/roamify-designing-and-evaluating-an-llm-based","title":"Roamify: Designing and Evaluating an LLM Based Google Chrome Extension for Personalised Itinerary Planning","date":"2025-03-10","arxiv_id":"2504.10489","n_code_links":1,"syntology":null},{"paper":null,"title":"Seeing Delta Parameters as JPEG Images: Data-Free Delta Compression with Discrete Cosine Transform","date":"2025-03-09","arxiv_id":"2503.06676","n_code_links":0,"syntology":null},{"paper":null,"title":"MoEMoE: Question Guided Dense and Scalable Sparse Mixture-of-Expert for Multi-source Multi-modal Answering","date":"2025-03-08","arxiv_id":"2503.06296","n_code_links":0,"syntology":null},{"paper":null,"title":"A Transformer Model for Predicting Chemical Reaction Products from Generic Templates","date":"2025-03-04","arxiv_id":"2503.05810","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":123},{"task":"/task/language-modeling","name":"Language Modeling","papers":96},{"task":"/task/question-answering","name":"Question Answering","papers":84},{"task":"/task/decoder","name":"Decoder","papers":78},{"task":"/task/text-generation","name":"Text Generation","papers":65},{"task":"/task/sentence","name":"Sentence","papers":55},{"task":"/task/translation","name":"Translation","papers":45},{"task":"/task/retrieval","name":"Retrieval","papers":40},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":40},{"task":"/task/machine-translation","name":"Machine Translation","papers":39},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":29},{"task":"/task/semantic-parsing","name":"Semantic Parsing","papers":23},{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":22},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":22},{"task":"/task/natural-language-inference","name":"Natural Language Inference","papers":20},{"task":"/task/code-generation","name":"Code Generation","papers":19},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":19},{"task":"/task/diversity","name":"Diversity","papers":18},{"task":"/task/text-summarization","name":"Text Summarization","papers":18},{"task":"/task/knowledge-graphs","name":"Knowledge Graphs","papers":17}],"tasks_shown":20,"n_tasks":467,"usage_by_year":[{"year":"2015","papers":1},{"year":"2019","papers":2},{"year":"2020","papers":30},{"year":"2021","papers":108},{"year":"2022","papers":163},{"year":"2023","papers":197},{"year":"2024","papers":151},{"year":"2025","papers":50}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/inverse-square-root-schedule"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}