{"url":"/method/gpt-neox","slug":"gpt-neox","name":"GPT-NeoX","full_name":"GPT-NeoX","full_name_withheld":false,"description_markdown":"**GPT-NeoX** is an autoregressive transformer decoder model whose architecture largely follows that of GPT-3, with a few notable deviations. The model has 20 billion parameters with 44 layers, a hidden dimension size of 6144, and 64 heads. The main difference with GPT-3 is the change in tokenizer, the addition of Rotary Positional Embeddings, the parallel computation of attention and feed-forward layers, and a different initialization scheme and hyperparameters.","description_state":"present","introduced_year":null,"introduced_by":{"title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","paper":"/paper/gpt-neox-20b-an-open-source-autoregressive-1","first_author":"Sid Black","n_authors":17,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/gpt-neox-20b-an-open-source-autoregressive-1"},"source":{"url":"https://arxiv.org/abs/2204.06745v1","title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":11,"archive_num_papers":11,"papers_newest_first":[{"paper":"/paper/extending-llms-context-window-with-100","title":"Extending LLMs' Context Window with 100 Samples","date":"2024-01-13","arxiv_id":"2401.07004","n_code_links":1,"syntology":{"ran":3,"of":4,"unverified":1,"pointer_only":4}},{"paper":"/paper/efficient-llm-inference-on-cpus","title":"Efficient LLM Inference on CPUs","date":"2023-11-01","arxiv_id":"2311.00502","n_code_links":2,"syntology":{"ran":2,"of":8,"unverified":6,"pointer_only":0}},{"paper":"/paper/clex-continuous-length-extrapolation-for","title":"CLEX: Continuous Length Extrapolation for Large Language Models","date":"2023-10-25","arxiv_id":"2310.16450","n_code_links":1,"syntology":{"ran":9,"of":10,"unverified":1,"pointer_only":0}},{"paper":null,"title":"How well can machine-generated texts be identified and can language models be trained to avoid identification?","date":"2023-10-25","arxiv_id":"2310.16992","n_code_links":0,"syntology":null},{"paper":"/paper/h2o-heavy-hitter-oracle-for-efficient","title":"H2O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/struc-bench-are-large-language-models-really","title":"Struc-Bench: Are Large Language Models Really Good at Generating Complex Structured Data?","date":"2023-09-16","arxiv_id":"2309.08963","n_code_links":1,"syntology":null},{"paper":"/paper/h-2-o-heavy-hitter-oracle-for-efficient","title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","date":"2023-06-24","arxiv_id":"2306.14048","n_code_links":2,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/goat-fine-tuned-llama-outperforms-gpt-4-on","title":"Goat: Fine-tuned LLaMA Outperforms GPT-4 on Arithmetic Tasks","date":"2023-05-23","arxiv_id":"2305.14201","n_code_links":1,"syntology":null},{"paper":"/paper/detectgpt-zero-shot-machine-generated-text","title":"DetectGPT: Zero-Shot Machine-Generated Text Detection using Probability Curvature","date":"2023-01-26","arxiv_id":"2301.11305","n_code_links":4,"syntology":{"ran":3,"of":9,"unverified":6,"pointer_only":2}},{"paper":"/paper/mass-editing-memory-in-a-transformer","title":"Mass-Editing Memory in a Transformer","date":"2022-10-13","arxiv_id":"2210.07229","n_code_links":2,"syntology":{"ran":6,"of":8,"unverified":2,"pointer_only":0}},{"paper":"/paper/gpt-neox-20b-an-open-source-autoregressive-1","title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","date":"2022-04-14","arxiv_id":"2204.06745","n_code_links":11,"syntology":{"ran":0,"of":3,"unverified":3,"pointer_only":0}}],"papers_shown":11,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":null,"name":"GPU","papers":2},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":null,"name":"Position","papers":2},{"task":"/task/4k","name":"4k","papers":1},{"task":"/task/articles","name":"Articles","papers":1},{"task":"/task/attribute","name":"Attribute","papers":1},{"task":"/task/dataset-generation","name":"Dataset Generation","papers":1},{"task":"/task/hallucination","name":"Hallucination","papers":1},{"task":"/task/linguistic-acceptability","name":"Linguistic Acceptability","papers":1},{"task":"/task/multi-task-language-understanding","name":"Multi-task Language Understanding","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/text-detection","name":"Text Detection","papers":1},{"task":"/task/text-generation","name":"Text Generation","papers":1}],"tasks_shown":14,"n_tasks":14,"usage_by_year":[{"year":"2022","papers":2},{"year":"2023","papers":8},{"year":"2024","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/gpt-neox"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}