{"url":"/method/gpt","slug":"gpt","name":"GPT","full_name":"GPT","full_name_withheld":false,"description_markdown":"**GPT** is a [Transformer](https://paperswithcode.com/method/transformer)-based architecture and training procedure for natural language processing tasks. Training follows a two-stage procedure. First, a language modeling objective is used on\r\nthe unlabeled data to learn the initial parameters of a neural network model. Subsequently, these parameters are adapted to a target task using the corresponding supervised objective.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Improving Language Understanding by Generative Pre-Training","paper":"/paper/improving-language-understanding-by","first_author":"Alec Radford","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/improving-language-understanding-by"},"source":{"url":"https://s3-us-west-2.amazonaws.com/openai-assets/research-covers/language-unsupervised/language_understanding_paper.pdf","title":"Improving Language Understanding by Generative Pre-Training","url_on_a_paper_host":false},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Autoregressive Transformers","url":"/methods/category/autoregressive-transformers","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":1212,"archive_num_papers":1212,"papers_newest_first":[{"paper":"/paper/making-language-model-a-hierarchical","title":"Making Language Model a Hierarchical Classifier and Generator","date":"2025-07-17","arxiv_id":"2507.12930","n_code_links":1,"syntology":null},{"paper":null,"title":"Generative Click-through Rate Prediction with Applications to Search Advertising","date":"2025-07-15","arxiv_id":"2507.11246","n_code_links":0,"syntology":null},{"paper":null,"title":"Behaviour Space Analysis of LLM-driven Meta-heuristic Discovery","date":"2025-07-04","arxiv_id":"2507.03605","n_code_links":0,"syntology":null},{"paper":null,"title":"Agent-to-Agent Theory of Mind: Testing Interlocutor Awareness among Large Language Models","date":"2025-06-28","arxiv_id":"2506.22957","n_code_links":0,"syntology":null},{"paper":null,"title":"Cat and Mouse -- Can Fake Text Generation Outpace Detector Systems?","date":"2025-06-26","arxiv_id":"2506.21274","n_code_links":0,"syntology":null},{"paper":null,"title":"Large Language Models Acing Chartered Accountancy","date":"2025-06-26","arxiv_id":"2506.21031","n_code_links":0,"syntology":null},{"paper":null,"title":"Large Language Model-Driven Code Compliance Checking in Building Information Modeling","date":"2025-06-25","arxiv_id":"2506.20551","n_code_links":0,"syntology":null},{"paper":null,"title":"InsertRank: LLMs can reason over BM25 scores to Improve Listwise Reranking","date":"2025-06-17","arxiv_id":"2506.14086","n_code_links":0,"syntology":null},{"paper":null,"title":"Toward a Graph Foundation Model: Pre-Training Transformers With Random Walks","date":"2025-06-17","arxiv_id":"2506.14098","n_code_links":0,"syntology":null},{"paper":"/paper/neuralnexus-at-bea-2025-shared-task-retrieval","title":"NeuralNexus at BEA 2025 Shared Task: Retrieval-Augmented Prompting for Mistake Identification in AI Tutors","date":"2025-06-12","arxiv_id":"2506.10627","n_code_links":1,"syntology":null},{"paper":null,"title":"Latent Multi-Head Attention for Small Language Models","date":"2025-06-11","arxiv_id":"2506.09342","n_code_links":0,"syntology":null},{"paper":null,"title":"AraReasoner: Evaluating Reasoning-Based LLMs for Arabic NLP","date":"2025-06-10","arxiv_id":"2506.08768","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-llms-across-multi-cognitive-levels","title":"Evaluating LLMs Across Multi-Cognitive Levels: From Medical Knowledge Mastery to Scenario-Based Problem Solving","date":"2025-06-10","arxiv_id":"2506.08349","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":3}},{"paper":null,"title":"Generative Voice Bursts during Phone Call","date":"2025-06-09","arxiv_id":"2506.07526","n_code_links":0,"syntology":null},{"paper":null,"title":"LLM-driven Indoor Scene Layout Generation via Scaled Human-aligned Data Synthesis and Multi-Stage Preference Optimization","date":"2025-06-09","arxiv_id":"2506.07570","n_code_links":0,"syntology":null},{"paper":null,"title":"RoboCerebra: A Large-scale Benchmark for Long-horizon Robotic Manipulation Evaluation","date":"2025-06-07","arxiv_id":"2506.06677","n_code_links":0,"syntology":null},{"paper":null,"title":"Evolutionary Perspectives on the Evaluation of LLM-Based AI Agents: A Comprehensive Survey","date":"2025-06-06","arxiv_id":"2506.11102","n_code_links":0,"syntology":null},{"paper":null,"title":"The Lock-in Hypothesis: Stagnation by Algorithm","date":"2025-06-06","arxiv_id":"2506.06166","n_code_links":0,"syntology":null},{"paper":"/paper/mathematical-reasoning-for-unmanned-aerial","title":"Mathematical Reasoning for Unmanned Aerial Vehicles: A RAG-Based Approach for Complex Arithmetic Reasoning","date":"2025-06-05","arxiv_id":"2506.04998","n_code_links":1,"syntology":null},{"paper":null,"title":"Privacy and Security Threat for OpenAI GPTs","date":"2025-06-04","arxiv_id":"2506.04036","n_code_links":0,"syntology":null},{"paper":"/paper/agent-x-evaluating-deep-multimodal-reasoning","title":"Agent-X: Evaluating Deep Multimodal Reasoning in Vision-Centric Agentic Tasks","date":"2025-05-30","arxiv_id":"2505.24876","n_code_links":1,"syntology":null},{"paper":"/paper/mofgpt-generative-design-of-metal-organic","title":"MOFGPT: Generative Design of Metal-Organic Frameworks using Language Models","date":"2025-05-30","arxiv_id":"2506.00198","n_code_links":1,"syntology":null},{"paper":null,"title":"When GPT Spills the Tea: Comprehensive Assessment of Knowledge File Leakage in GPTs","date":"2025-05-30","arxiv_id":"2506.00197","n_code_links":0,"syntology":null},{"paper":null,"title":"Daunce: Data Attribution through Uncertainty Estimation","date":"2025-05-29","arxiv_id":"2505.23223","n_code_links":0,"syntology":null},{"paper":null,"title":"Reducing Latency in LLM-Based Natural Language Commands Processing for Robot Navigation","date":"2025-05-29","arxiv_id":"2506.00075","n_code_links":0,"syntology":null},{"paper":null,"title":"Multi-MLLM Knowledge Distillation for Out-of-Context News Detection","date":"2025-05-28","arxiv_id":"2505.22517","n_code_links":0,"syntology":null},{"paper":"/paper/explainability-of-large-language-models-using","title":"Explainability of Large Language Models using SMILE: Statistical Model-agnostic Interpretability with Local Explanations","date":"2025-05-27","arxiv_id":"2505.21657","n_code_links":1,"syntology":null},{"paper":null,"title":"From prosthetic memory to prosthetic denial: Auditing whether large language models are prone to mass atrocity denialism","date":"2025-05-27","arxiv_id":"2505.21753","n_code_links":0,"syntology":null},{"paper":null,"title":"Automated evaluation of children's speech fluency for low-resource languages","date":"2025-05-26","arxiv_id":"2505.19671","n_code_links":0,"syntology":null},{"paper":null,"title":"Beyond Specialization: Benchmarking LLMs for Transliteration of Indian Languages","date":"2025-05-26","arxiv_id":"2505.19851","n_code_links":0,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":193},{"task":"/task/language-modeling","name":"Language Modeling","papers":138},{"task":"/task/large-language-model","name":"Large Language Model","papers":98},{"task":"/task/question-answering","name":"Question Answering","papers":71},{"task":"/task/text-generation","name":"Text Generation","papers":68},{"task":"/task/retrieval","name":"Retrieval","papers":57},{"task":"/task/prompt-engineering","name":"Prompt Engineering","papers":52},{"task":"/task/sentence","name":"Sentence","papers":50},{"task":"/task/decoder","name":"Decoder","papers":45},{"task":"/task/decision-making","name":"Decision Making","papers":38},{"task":"/task/translation","name":"Translation","papers":35},{"task":"/task/in-context-learning","name":"In-Context Learning","papers":34},{"task":"/task/text-classification","name":"Text Classification","papers":34},{"task":"/task/text-classification-1","name":"text-classification","papers":33},{"task":"/task/articles","name":"Articles","papers":31},{"task":"/task/few-shot-learning","name":"Few-Shot Learning","papers":31},{"task":"/task/rag","name":"RAG","papers":31},{"task":"/task/retrieval-augmented-generation","name":"Retrieval-augmented Generation","papers":31},{"task":"/task/fairness","name":"Fairness","papers":30},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":29}],"tasks_shown":20,"n_tasks":583,"usage_by_year":[{"year":"2018","papers":1},{"year":"2019","papers":28},{"year":"2020","papers":23},{"year":"2021","papers":50},{"year":"2022","papers":62},{"year":"2023","papers":360},{"year":"2024","papers":515},{"year":"2025","papers":173}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/gpt"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}