{"url":"/method/pythia","slug":"pythia","name":"Pythia","full_name":"Pythia","full_name_withheld":false,"description_markdown":"**Pythia** is a suite of decoder-only autoregressive language models all trained on public data seen in the exact same order and ranging in size from 70M to 12B parameters. The model architecture and hyperparameters largely follow GPT-3, with a few notable deviations based on recent advances in best practices for large scale language modeling.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","paper":"/paper/pythia-a-suite-for-analyzing-large-language","first_author":"Stella Biderman","n_authors":13,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/pythia-a-suite-for-analyzing-large-language"},"source":{"url":"https://arxiv.org/abs/2304.01373v2","title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":60,"archive_num_papers":60,"papers_newest_first":[{"paper":"/paper/leximark-robust-watermarking-via-lexical","title":"LexiMark: Robust Watermarking via Lexical Substitutions to Enhance Membership Verification of an LLM's Textual Training Data","date":"2025-06-17","arxiv_id":"2506.14474","n_code_links":1,"syntology":null},{"paper":"/paper/what-happens-during-the-loss-plateau","title":"What Happens During the Loss Plateau? Understanding Abrupt Learning in Transformers","date":"2025-06-16","arxiv_id":"2506.13688","n_code_links":1,"syntology":{"ran":0,"of":2,"unverified":2,"pointer_only":2}},{"paper":null,"title":"Stochastic Chameleons: Irrelevant Context Hallucinations Reveal Class-Based (Mis)Generalization in LLMs","date":"2025-05-28","arxiv_id":"2505.22630","n_code_links":0,"syntology":null},{"paper":"/paper/pretraining-language-models-to-ponder-in","title":"Pretraining Language Models to Ponder in Continuous Space","date":"2025-05-27","arxiv_id":"2505.20674","n_code_links":1,"syntology":{"ran":6,"of":6,"unverified":0,"pointer_only":3}},{"paper":"/paper/2505-11004","title":"Illusion or Algorithm? Investigating Memorization, Emergence, and Symbolic Processing in In-Context Learning","date":"2025-05-16","arxiv_id":"2505.11004","n_code_links":1,"syntology":null},{"paper":null,"title":"Memorization or Interpolation ? Detecting LLM Memorization through Input Perturbation Analysis","date":"2025-05-05","arxiv_id":"2505.03019","n_code_links":0,"syntology":null},{"paper":null,"title":"An Empirical Study of the Role of Incompleteness and Ambiguity in Interactions with Large Language Models","date":"2025-03-23","arxiv_id":"2503.17936","n_code_links":0,"syntology":null},{"paper":null,"title":"I Predict Therefore I Am: Is Next Token Prediction Enough to Learn Human-Interpretable Concepts from Data?","date":"2025-03-12","arxiv_id":"2503.08980","n_code_links":0,"syntology":null},{"paper":"/paper/polypythias-stability-and-outliers-across","title":"PolyPythias: Stability and Outliers across Fifty Language Model Pre-Training Runs","date":"2025-03-12","arxiv_id":"2503.09543","n_code_links":1,"syntology":null},{"paper":null,"title":"Interrogating LLM design under a fair learning doctrine","date":"2025-02-22","arxiv_id":"2502.16290","n_code_links":0,"syntology":null},{"paper":null,"title":"Revisiting Privacy, Utility, and Efficiency Trade-offs when Fine-Tuning Large Language Models","date":"2025-02-18","arxiv_id":"2502.13313","n_code_links":0,"syntology":null},{"paper":"/paper/roste-an-efficient-quantization-aware","title":"RoSTE: An Efficient Quantization-Aware Supervised Fine-Tuning Approach for Large Language Models","date":"2025-02-13","arxiv_id":"2502.09003","n_code_links":0,"syntology":{"ran":5,"of":5,"unverified":0,"pointer_only":5}},{"paper":null,"title":"MemHunter: Automated and Verifiable Memorization Detection at Dataset-scale in LLMs","date":"2024-12-10","arxiv_id":"2412.07261","n_code_links":0,"syntology":null},{"paper":"/paper/star-agents-automatic-data-optimization-with","title":"Star-Agents: Automatic Data Optimization with LLM Agents for Instruction Tuning","date":"2024-11-21","arxiv_id":"2411.14497","n_code_links":1,"syntology":null},{"paper":"/paper/explaining-and-improving-contrastive-decoding","title":"Explaining and Improving Contrastive Decoding by Extrapolating the Probabilities of a Huge and Hypothetical LM","date":"2024-11-03","arxiv_id":"2411.01610","n_code_links":1,"syntology":null},{"paper":null,"title":"Efficient Training of Sparse Autoencoders for Large Language Models via Layer Groups","date":"2024-10-28","arxiv_id":"2410.21508","n_code_links":0,"syntology":null},{"paper":null,"title":"Relaxed Recursive Transformers: Effective Parameter Sharing with Layer-wise LoRA","date":"2024-10-28","arxiv_id":"2410.20672","n_code_links":0,"syntology":null},{"paper":null,"title":"Hallucination Detox: Sensitivity Dropout (SenD) for Large Language Model Training","date":"2024-10-20","arxiv_id":"2410.15460","n_code_links":0,"syntology":null},{"paper":"/paper/tending-towards-stability-convergence","title":"Tending Towards Stability: Convergence Challenges in Small Language Models","date":"2024-10-15","arxiv_id":"2410.11451","n_code_links":1,"syntology":null},{"paper":null,"title":"Context-Parametric Inversion: Why Instruction Finetuning Can Worsen Context Reliance","date":"2024-10-14","arxiv_id":"2410.10796","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-model-evaluation-via-matrix-1","title":"Large Language Model Evaluation via Matrix Nuclear-Norm","date":"2024-10-14","arxiv_id":"2410.10672","n_code_links":1,"syntology":null},{"paper":"/paper/local-and-global-decoding-in-text-generation","title":"Local and Global Decoding in Text Generation","date":"2024-10-14","arxiv_id":"2410.10810","n_code_links":1,"syntology":{"ran":11,"of":12,"unverified":1,"pointer_only":12}},{"paper":"/paper/reconstruction-of-particle-flow-energy","title":"Lightweight Deep Learning Framework for Accurate Particle Flow Energy Reconstruction","date":"2024-10-08","arxiv_id":"2410.07250","n_code_links":1,"syntology":null},{"paper":null,"title":"Order of Magnitude Speedups for LLM Membership Inference","date":"2024-09-22","arxiv_id":"2409.14513","n_code_links":0,"syntology":null},{"paper":null,"title":"Generated Data with Fake Privacy: Hidden Dangers of Fine-tuning Large Language Models on Generated Data","date":"2024-09-12","arxiv_id":"2409.11423","n_code_links":0,"syntology":null},{"paper":null,"title":"Accelerating Large Language Model Pretraining via LFR Pedagogy: Learn, Focus, and Review","date":"2024-09-10","arxiv_id":"2409.06131","n_code_links":0,"syntology":null},{"paper":"/paper/demystifying-verbatim-memorization-in-large","title":"Demystifying Verbatim Memorization in Large Language Models","date":"2024-07-25","arxiv_id":"2407.17817","n_code_links":1,"syntology":{"ran":6,"of":6,"unverified":0,"pointer_only":0}},{"paper":"/paper/generalization-v-s-memorization-tracing","title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","date":"2024-07-20","arxiv_id":"2407.14985","n_code_links":0,"syntology":{"ran":11,"of":15,"unverified":4,"pointer_only":0}},{"paper":"/paper/mathcamps-fine-grained-synthesis-of","title":"MathCAMPS: Fine-grained Synthesis of Mathematical Problems From Human Curricula","date":"2024-07-01","arxiv_id":"2407.00900","n_code_links":1,"syntology":{"ran":8,"of":8,"unverified":0,"pointer_only":0}},{"paper":"/paper/evaluating-n-gram-novelty-of-language-models","title":"Evaluating $n$-Gram Novelty of Language Models Using Rusty-DAWG","date":"2024-06-18","arxiv_id":"2406.13069","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":23},{"task":"/task/language-modeling","name":"Language Modeling","papers":15},{"task":"/task/memorization","name":"Memorization","papers":10},{"task":"/task/large-language-model","name":"Large Language Model","papers":5},{"task":"/task/question-answering","name":"Question Answering","papers":4},{"task":"/task/text-generation","name":"Text Generation","papers":3},{"task":"/task/articles","name":"Articles","papers":2},{"task":"/task/common-sense-reasoning","name":"Common Sense Reasoning","papers":2},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":2},{"task":"/task/hallucination","name":"Hallucination","papers":2},{"task":"/task/in-context-learning","name":"In-Context Learning","papers":2},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":2},{"task":"/task/lambada","name":"LAMBADA","papers":2},{"task":"/task/machine-translation","name":"Machine Translation","papers":2},{"task":"/task/math","name":"Math","papers":2},{"task":"/task/sentence-completion","name":"Sentence Completion","papers":2},{"task":"/task/world-knowledge","name":"World Knowledge","papers":2},{"task":null,"name":"counterfactual","papers":2},{"task":"/task/model","name":"model","papers":2},{"task":"/task/adversarial-attack","name":"Adversarial Attack","papers":1}],"tasks_shown":20,"n_tasks":58,"usage_by_year":[{"year":"2023","papers":17},{"year":"2024","papers":31},{"year":"2025","papers":12}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/pythia"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}