{"url":"/method/prophetnet","slug":"prophetnet","name":"ProphetNet","full_name":"ProphetNet","full_name_withheld":false,"description_markdown":"**ProphetNet** is a sequence-to-sequence pre-training model that introduces a novel self-supervised objective named future n-gram prediction and the proposed n-stream self-attention mechanism. Instead of optimizing one-step-ahead prediction in the traditional sequence-to-sequence model, the ProphetNet is optimized by $n$-step ahead prediction that predicts the next $n$ tokens simultaneously based on previous context tokens at each time step. The future n-gram prediction explicitly encourages the model to plan for the future tokens and further help predict multiple future tokens.","description_state":"present","introduced_year":null,"introduced_by":{"title":"ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training","paper":"/paper/prophetnet-predicting-future-n-gram-for","first_author":"Weizhen Qi","n_authors":8,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/prophetnet-predicting-future-n-gram-for"},"source":{"url":"https://arxiv.org/abs/2001.04063v3","title":"ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]},{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":11,"archive_num_papers":11,"papers_newest_first":[{"paper":"/paper/sats-simplification-aware-text-summarization","title":"SATS: simplification aware text summarization of scientific documents","date":"2024-07-10","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Evaluating Text Summaries Generated by Large Language Models Using OpenAI's GPT","date":"2024-05-07","arxiv_id":"2405.04053","n_code_links":0,"syntology":null},{"paper":null,"title":"Analysis of Multidomain Abstractive Summarization Using Salience Allocation","date":"2024-02-19","arxiv_id":"2402.11955","n_code_links":0,"syntology":null},{"paper":null,"title":"Predicting Temperature of Major Cities Using Machine Learning and Deep Learning","date":"2023-09-23","arxiv_id":"2309.13330","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-pre-trained-models-with-text","title":"Enhancing Pre-trained Models with Text Structure Knowledge for Question Generation","date":"2022-09-09","arxiv_id":"2209.04179","n_code_links":0,"syntology":null},{"paper":"/paper/prophetnet-x-large-scale-pre-training-models","title":"ProphetNet-X: Large-Scale Pre-training Models for English, Chinese, Multi-lingual, Dialog, and Code Generation","date":"2021-04-16","arxiv_id":"2104.08006","n_code_links":3,"syntology":null},{"paper":null,"title":"A Survey of Recent Abstract Summarization Techniques","date":"2021-04-15","arxiv_id":"2105.00824","n_code_links":0,"syntology":null},{"paper":"/paper/glge-a-new-general-language-generation","title":"GLGE: A New General Language Generation Evaluation Benchmark","date":"2020-11-24","arxiv_id":"2011.11928","n_code_links":1,"syntology":null},{"paper":"/paper/prophetnet-predicting-future-n-gram-for-1","title":"ProphetNet: Predicting Future N-gram for Sequence-to-SequencePre-training","date":"2020-11-01","arxiv_id":null,"n_code_links":3,"syntology":null},{"paper":"/paper/topic-aware-abstractive-text-summarization","title":"Topic-Guided Abstractive Text Summarization: a Joint Learning Approach","date":"2020-10-20","arxiv_id":"2010.10323","n_code_links":1,"syntology":null},{"paper":"/paper/prophetnet-predicting-future-n-gram-for","title":"ProphetNet: Predicting Future N-gram for Sequence-to-Sequence Pre-training","date":"2020-01-13","arxiv_id":"2001.04063","n_code_links":5,"syntology":null}],"papers_shown":11,"tasks":[{"task":"/task/text-summarization","name":"Text Summarization","papers":5},{"task":"/task/abstractive-text-summarization","name":"Abstractive Text Summarization","papers":4},{"task":"/task/question-generation","name":"Question Generation","papers":4},{"task":"/task/question-generation","name":"Question-Generation","papers":4},{"task":"/task/prediction","name":"Prediction","papers":3},{"task":"/task/text-generation","name":"Text Generation","papers":2},{"task":"/task/articles","name":"Articles","papers":1},{"task":"/task/code-generation","name":"Code Generation","papers":1},{"task":"/task/event-driven-trading","name":"Event-Driven Trading","papers":1},{"task":"/task/extractive-summarization","name":"Extractive Summarization","papers":1},{"task":"/task/future-prediction","name":"Future prediction","papers":1},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":1},{"task":"/task/open-domain-dialog","name":"Open-Domain Dialog","papers":1},{"task":"/task/survey","name":"Survey","papers":1},{"task":"/task/text-simplification","name":"Text Simplification","papers":1},{"task":"/task/time-series","name":"Time Series Analysis","papers":1},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":1},{"task":"/task/translation","name":"Translation","papers":1}],"tasks_shown":18,"n_tasks":18,"usage_by_year":[{"year":"2020","papers":4},{"year":"2021","papers":2},{"year":"2022","papers":1},{"year":"2023","papers":1},{"year":"2024","papers":3}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/prophetnet"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}