{"url":"/method/gpt-neo","slug":"gpt-neo","name":"GPT-Neo","full_name":"GPT-Neo","full_name_withheld":false,"description_markdown":"An implementation of model & data parallel [GPT3-like](https://paperswithcode.com/method/gpt-3) models using the [mesh-tensorflow](https://github.com/tensorflow/mesh) library.\r\n\r\nSource: [EleutherAI/GPT-Neo](https://github.com/EleutherAI/gpt-neo)","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":null,"title":null,"url_on_a_paper_host":false},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":38,"archive_num_papers":38,"papers_newest_first":[{"paper":null,"title":"IRepair: An Intent-Aware Approach to Repair Data-Driven Errors in Large Language Models","date":"2025-02-10","arxiv_id":"2502.07072","n_code_links":0,"syntology":null},{"paper":"/paper/robust-hybrid-classical-quantum-transfer","title":"Robust Hybrid Classical-Quantum Transfer Learning Model for Text Classification Using GPT-Neo 125M with LoRA & SMOTE Enhancement","date":"2025-01-12","arxiv_id":"2501.10435","n_code_links":1,"syntology":null},{"paper":null,"title":"LLM Vocabulary Compression for Low-Compute Environments","date":"2024-11-10","arxiv_id":"2411.06371","n_code_links":0,"syntology":null},{"paper":"/paper/berttime-stories-investigating-the-role-of","title":"BERTtime Stories: Investigating the Role of Synthetic Story Data in Language pre-training","date":"2024-10-20","arxiv_id":"2410.15365","n_code_links":1,"syntology":null},{"paper":null,"title":"Reconstruction of Differentially Private Text Sanitization via Large Language Models","date":"2024-10-16","arxiv_id":"2410.12443","n_code_links":0,"syntology":null},{"paper":"/paper/the-unreasonable-ineffectiveness-of-nucleus","title":"The Unreasonable Ineffectiveness of Nucleus Sampling on Mitigating Text Memorization","date":"2024-08-29","arxiv_id":"2408.16345","n_code_links":1,"syntology":{"ran":8,"of":11,"unverified":3,"pointer_only":11}},{"paper":null,"title":"WPN: An Unlearning Method Based on N-pair Contrastive Learning in Language Models","date":"2024-08-18","arxiv_id":"2408.09459","n_code_links":0,"syntology":null},{"paper":"/paper/towards-robust-and-cost-efficient-knowledge","title":"Towards Robust and Parameter-Efficient Knowledge Unlearning for LLMs","date":"2024-08-13","arxiv_id":"2408.06621","n_code_links":1,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":0}},{"paper":null,"title":"Semantic Membership Inference Attack against Large Language Models","date":"2024-06-14","arxiv_id":"2406.10218","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-wit-creativity-and","title":"Investigating Wit, Creativity, and Detectability of Large Language Models in Domain-Specific Writing Style Adaptation of Reddit's Showerthoughts","date":"2024-05-02","arxiv_id":"2405.01660","n_code_links":1,"syntology":null},{"paper":null,"title":"More than Correlation: Do Large Language Models Learn Causal Representations of Space?","date":"2023-12-26","arxiv_id":"2312.16257","n_code_links":0,"syntology":null},{"paper":"/paper/fairness-aware-structured-pruning-in","title":"Fairness-Aware Structured Pruning in Transformers","date":"2023-12-24","arxiv_id":"2312.15398","n_code_links":1,"syntology":{"ran":0,"of":2,"unverified":2,"pointer_only":0}},{"paper":null,"title":"Scalable Extraction of Training Data from (Production) Language Models","date":"2023-11-28","arxiv_id":"2311.17035","n_code_links":0,"syntology":null},{"paper":"/paper/heaps-law-in-gpt-neo-large-language-model","title":"Heaps' Law in GPT-Neo Large Language Model Emulated Corpora","date":"2023-11-10","arxiv_id":"2311.06377","n_code_links":1,"syntology":null},{"paper":"/paper/watermarking-llms-with-weight-quantization","title":"Watermarking LLMs with Weight Quantization","date":"2023-10-17","arxiv_id":"2310.11237","n_code_links":1,"syntology":{"ran":1,"of":10,"unverified":9,"pointer_only":10}},{"paper":"/paper/tart-a-plug-and-play-transformer-module-for","title":"TART: A plug-and-play Transformer module for task-agnostic reasoning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/fine-tuning-large-language-models-for","title":"Fine-Tuning Large Language Models for Answering Programming Questions with Code Snippets","date":"2023-06-26","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Exposing Bias in Online Communities through Large-Scale Language Models","date":"2023-06-04","arxiv_id":"2306.02294","n_code_links":0,"syntology":null},{"paper":"/paper/test-time-training-on-nearest-neighbors-for","title":"Test-Time Training on Nearest Neighbors for Large Language Models","date":"2023-05-29","arxiv_id":"2305.18466","n_code_links":1,"syntology":{"ran":0,"of":5,"unverified":5,"pointer_only":0}},{"paper":"/paper/controlling-the-extraction-of-memorized-data","title":"Controlling the Extraction of Memorized Data from Large Language Models via Prompt-Tuning","date":"2023-05-19","arxiv_id":"2305.11759","n_code_links":1,"syntology":null},{"paper":"/paper/tinystories-how-small-can-language-models-be","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","date":"2023-05-12","arxiv_id":"2305.07759","n_code_links":8,"syntology":{"ran":3,"of":18,"unverified":15,"pointer_only":0}},{"paper":"/paper/on-the-risks-of-stealing-the-decoding","title":"Stealing the Decoding Algorithms of Language Models","date":"2023-03-08","arxiv_id":"2303.04729","n_code_links":1,"syntology":null},{"paper":"/paper/bag-of-tricks-for-training-data-extraction","title":"Bag of Tricks for Training Data Extraction from Language Models","date":"2023-02-09","arxiv_id":"2302.04460","n_code_links":1,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":1}},{"paper":null,"title":"Why Does Surprisal From Larger Transformer-Based Language Models Provide a Poorer Fit to Human Reading Times?","date":"2022-12-23","arxiv_id":"2212.12131","n_code_links":0,"syntology":null},{"paper":null,"title":"Explicit Knowledge Transfer for Weakly-Supervised Code Generation","date":"2022-11-30","arxiv_id":"2211.16740","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-neo-for-commonsense-reasoning-a","title":"GPT-Neo for commonsense reasoning -- a theoretical and practical lens","date":"2022-11-28","arxiv_id":"2211.15593","n_code_links":1,"syntology":null},{"paper":null,"title":"Understanding BLOOM: An empirical study on diverse NLP tasks","date":"2022-11-27","arxiv_id":"2211.14865","n_code_links":0,"syntology":null},{"paper":"/paper/collateral-facilitation-in-humans-and","title":"Collateral facilitation in humans and language models","date":"2022-11-09","arxiv_id":"2211.05198","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":null,"title":"Chain of Explanation: New Prompting Method to Generate Higher Quality Natural Language Explanation for Implicit Hate Speech","date":"2022-09-11","arxiv_id":"2209.04889","n_code_links":0,"syntology":null},{"paper":"/paper/materials-transformers-language-models-for","title":"Materials Transformers Language Models for Generative Materials Design: a benchmark study","date":"2022-06-27","arxiv_id":"2206.13578","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":10},{"task":"/task/text-generation","name":"Text Generation","papers":9},{"task":"/task/language-modeling","name":"Language Modeling","papers":8},{"task":"/task/code-generation","name":"Code Generation","papers":5},{"task":"/task/memorization","name":"Memorization","papers":3},{"task":"/task/prompt-engineering","name":"Prompt Engineering","papers":3},{"task":"/task/large-language-model","name":"Large Language Model","papers":2},{"task":"/task/question-answering","name":"Question Answering","papers":2},{"task":"/task/text-classification","name":"Text Classification","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2},{"task":"/task/text-classification-1","name":"text-classification","papers":2},{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":1},{"task":"/task/all","name":"All","papers":1},{"task":"/task/machine-learning","name":"BIG-bench Machine Learning","papers":1},{"task":"/task/chatbot","name":"Chatbot","papers":1},{"task":"/task/contrastive-learning","name":"Contrastive Learning","papers":1},{"task":"/task/coreference-resolution","name":"Coreference Resolution","papers":1},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/diagnostic","name":"Diagnostic","papers":1},{"task":"/task/fairness","name":"Fairness","papers":1}],"tasks_shown":20,"n_tasks":42,"usage_by_year":[{"year":"2021","papers":4},{"year":"2022","papers":11},{"year":"2023","papers":13},{"year":"2024","papers":8},{"year":"2025","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/gpt-neo"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}