{"url":"/method/gpt-3","slug":"gpt-3","name":"GPT-3","full_name":"GPT-3","full_name_withheld":false,"description_markdown":"**GPT-3** is an autoregressive [transformer](https://paperswithcode.com/methods/category/transformers)  model with 175 billion\r\nparameters. It uses the same architecture/model as [GPT-2](https://paperswithcode.com/method/gpt-2), including the modified initialization, pre-normalization, and reversible tokenization, with the exception that GPT-3 uses alternating dense and locally banded sparse attention patterns in the layers of the [transformer](https://paperswithcode.com/method/transformer), similar to the [Sparse Transformer](https://paperswithcode.com/method/sparse-transformer).","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/2005.14165v4","title":"Language Models are Few-Shot Learners","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/EleutherAI/gpt-neox","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":1906,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"Augmenting Large Language Models with Static Code Analysis for Automated Code Quality Improvements","date":"2025-06-12","arxiv_id":"2506.10330","n_code_links":0,"syntology":null},{"paper":"/paper/think-before-you-simulate-symbolic-reasoning-1","title":"Think before You Simulate: Symbolic Reasoning to Orchestrate Neural Computation for Counterfactual Question Answering","date":"2025-06-12","arxiv_id":"2506.10753","n_code_links":1,"syntology":null},{"paper":null,"title":"Multilingual Hate Speech Detection in Social Media Using Translation-Based Approaches with Large Language Models","date":"2025-06-09","arxiv_id":"2506.08147","n_code_links":0,"syntology":null},{"paper":null,"title":"Direct Behavior Optimization: Unlocking the Potential of Lightweight LLMs","date":"2025-06-06","arxiv_id":"2506.06401","n_code_links":0,"syntology":null},{"paper":null,"title":"Benchmarking Large Language Models on Homework Assessment in Circuit Analysis","date":"2025-06-05","arxiv_id":"2506.06390","n_code_links":0,"syntology":null},{"paper":null,"title":"Multiple-Choice Question Generation Using Large Language Models: Methodology and Educator Insights","date":"2025-06-05","arxiv_id":"2506.04851","n_code_links":0,"syntology":null},{"paper":null,"title":"Facts are Harder Than Opinions -- A Multilingual, Comparative Analysis of LLM-Based Fact-Checking Reliability","date":"2025-06-04","arxiv_id":"2506.03655","n_code_links":0,"syntology":null},{"paper":null,"title":"FinBERT2: A Specialized Bidirectional Encoder for Bridging the Gap in Finance-Specific Deployment of Large Language Models","date":"2025-05-31","arxiv_id":"2506.06335","n_code_links":0,"syntology":null},{"paper":null,"title":"Critical Batch Size Revisited: A Simple Empirical Approach to Large-Batch Language Model Training","date":"2025-05-29","arxiv_id":"2505.23971","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing LLM-Based Code Generation with Complexity Metrics: A Feedback-Driven Approach","date":"2025-05-29","arxiv_id":"2505.23953","n_code_links":0,"syntology":null},{"paper":null,"title":"Say What You Mean: Natural Language Access Control with Large Language Models for Internet of Things","date":"2025-05-28","arxiv_id":"2505.23835","n_code_links":0,"syntology":null},{"paper":null,"title":"Conversational Lexicography: Querying Lexicographic Data on Knowledge Graphs with SPARQL through Natural Language","date":"2025-05-26","arxiv_id":"2505.19971","n_code_links":0,"syntology":null},{"paper":"/paper/generative-ai-and-creativity-a-systematic","title":"Generative AI and Creativity: A Systematic Literature Review and Meta-Analysis","date":"2025-05-22","arxiv_id":"2505.17241","n_code_links":1,"syntology":null},{"paper":null,"title":"Adversarial Testing in LLMs: Insights into Decision-Making Vulnerabilities","date":"2025-05-19","arxiv_id":"2505.13195","n_code_links":0,"syntology":null},{"paper":null,"title":"Are Large Language Models Good at Detecting Propaganda?","date":"2025-05-19","arxiv_id":"2505.13706","n_code_links":0,"syntology":null},{"paper":null,"title":"EVALOOP: Assessing LLM Robustness in Programming from a Self-consistency Perspective","date":"2025-05-18","arxiv_id":"2505.12185","n_code_links":0,"syntology":null},{"paper":null,"title":"Let the Trial Begin: A Mock-Court Approach to Vulnerability Detection using LLM-Based Agents","date":"2025-05-16","arxiv_id":"2505.10961","n_code_links":0,"syntology":null},{"paper":"/paper/comparing-llm-text-annotation-skills-a-study","title":"Comparing LLM Text Annotation Skills: A Study on Human Rights Violations in Social Media Data","date":"2025-05-15","arxiv_id":"2505.10260","n_code_links":1,"syntology":null},{"paper":"/paper/achieving-scalable-robot-autonomy-via","title":"Achieving Scalable Robot Autonomy via neurosymbolic planning using lightweight local LLM","date":"2025-05-13","arxiv_id":"2505.08492","n_code_links":1,"syntology":null},{"paper":null,"title":"Evaluating the Effectiveness of Black-Box Prompt Optimization as the Scale of LLMs Continues to Grow","date":"2025-05-13","arxiv_id":"2505.08303","n_code_links":0,"syntology":null},{"paper":"/paper/healthbench-evaluating-large-language-models","title":"HealthBench: Evaluating Large Language Models Towards Improved Human Health","date":"2025-05-13","arxiv_id":"2505.08775","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":0}},{"paper":"/paper/grada-graph-based-reranker-against","title":"GRADA: Graph-based Reranker against Adversarial Documents Attack","date":"2025-05-12","arxiv_id":"2505.07546","n_code_links":1,"syntology":null},{"paper":null,"title":"REFINE-AF: A Task-Agnostic Framework to Align Language Models via Self-Generated Instructions using Reinforcement Learning from Automated Feedback","date":"2025-05-10","arxiv_id":"2505.06548","n_code_links":0,"syntology":null},{"paper":null,"title":"An empathic GPT-based chatbot to talk about mental disorders with Spanish teenagers","date":"2025-05-09","arxiv_id":"2505.05828","n_code_links":0,"syntology":null},{"paper":null,"title":"What Is Next for LLMs? Next-Generation AI Computing Hardware Using Photonic Chips","date":"2025-05-09","arxiv_id":"2505.05794","n_code_links":0,"syntology":null},{"paper":null,"title":"Performance Evaluation of Large Language Models in Bangla Consumer Health Query Summarization","date":"2025-05-08","arxiv_id":"2505.05070","n_code_links":0,"syntology":null},{"paper":null,"title":"Bringing legal knowledge to the public by constructing a legal question bank using large-scale pre-trained language model","date":"2025-05-07","arxiv_id":"2505.04132","n_code_links":0,"syntology":null},{"paper":null,"title":"A Domain Adaptation of Large Language Models for Classifying Mechanical Assembly Components","date":"2025-05-02","arxiv_id":"2505.01627","n_code_links":0,"syntology":null},{"paper":null,"title":"JaccDiv: A Metric and Benchmark for Quantifying Diversity of Generated Marketing Text in the Music Industry","date":"2025-04-29","arxiv_id":"2504.20849","n_code_links":0,"syntology":null},{"paper":null,"title":"VeriDebug: A Unified LLM for Verilog Debugging via Contrastive Embedding and Guided Correction","date":"2025-04-27","arxiv_id":"2504.19099","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":363},{"task":"/task/language-modeling","name":"Language Modeling","papers":265},{"task":"/task/question-answering","name":"Question Answering","papers":206},{"task":"/task/large-language-model","name":"Large Language Model","papers":187},{"task":"/task/retrieval","name":"Retrieval","papers":128},{"task":"/task/in-context-learning","name":"In-Context Learning","papers":115},{"task":"/task/text-generation","name":"Text Generation","papers":103},{"task":"/task/prompt-engineering","name":"Prompt Engineering","papers":88},{"task":"/task/code-generation","name":"Code Generation","papers":80},{"task":"/task/few-shot-learning","name":"Few-Shot Learning","papers":78},{"task":"/task/sentence","name":"Sentence","papers":75},{"task":"/task/math","name":"Math","papers":61},{"task":"/task/retrieval-augmented-generation","name":"Retrieval-augmented Generation","papers":59},{"task":"/task/rag","name":"RAG","papers":58},{"task":"/task/decision-making","name":"Decision Making","papers":54},{"task":"/task/multiple-choice","name":"Multiple-choice","papers":54},{"task":"/task/hallucination","name":"Hallucination","papers":45},{"task":"/task/benchmarking","name":"Benchmarking","papers":43},{"task":"/task/zero-shot-learning","name":"Zero-Shot Learning","papers":42},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":40}],"tasks_shown":20,"n_tasks":665,"usage_by_year":[{"year":"2020","papers":16},{"year":"2021","papers":99},{"year":"2022","papers":219},{"year":"2023","papers":744},{"year":"2024","papers":716},{"year":"2025","papers":112}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/gpt-3"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}