{"url":"/method/dapo","slug":"dapo","name":"DAPO","full_name":"Dialogue-Adaptive Pre-training Objective","full_name_withheld":false,"description_markdown":"**Dialogue-Adaptive Pre-training Objective (DAPO)** is a pre-training objective for dialogue adaptation, which is designed to measure qualities of dialogues from multiple important aspects, like Readability, Consistency and Fluency which have already been focused on by general LM pre-training objectives, and those also significant for assessing dialogues but ignored by general LM pre-training objectives, like Diversity and Specificity.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Dialogue-adaptive Language Model Pre-training From Quality Estimation","paper":"/paper/task-specific-objectives-of-pre-trained","first_author":"Junlong Li","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/task-specific-objectives-of-pre-trained"},"source":{"url":"https://arxiv.org/abs/2009.04984v2","title":"Dialogue-adaptive Language Model Pre-training From Quality Estimation","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Dialog Adaptation","url":"/methods/category/dialog-adaptation","pwc_aliases":[]}],"n_papers_tagged":11,"archive_num_papers":11,"papers_newest_first":[{"paper":null,"title":"MemAgent: Reshaping Long-Context LLM with Multi-Conv RL-based Memory Agent","date":"2025-07-03","arxiv_id":"2507.02259","n_code_links":0,"syntology":null},{"paper":null,"title":"CodeV-R1: Reasoning-Enhanced Verilog Generation","date":"2025-05-30","arxiv_id":"2505.24183","n_code_links":0,"syntology":null},{"paper":null,"title":"CoThink: Token-Efficient Reasoning via Instruct Models Guiding Reasoning Models","date":"2025-05-28","arxiv_id":"2505.22017","n_code_links":0,"syntology":null},{"paper":"/paper/masksearch-a-universal-pre-training-framework","title":"MASKSEARCH: A Universal Pre-Training Framework to Enhance Agentic Search Capability","date":"2025-05-26","arxiv_id":"2505.20285","n_code_links":1,"syntology":null},{"paper":null,"title":"Rethinking the Sampling Criteria in Reinforcement Learning for LLM Reasoning: A Competence-Difficulty Alignment Perspective","date":"2025-05-23","arxiv_id":"2505.17652","n_code_links":0,"syntology":null},{"paper":"/paper/ktae-a-model-free-algorithm-to-key-tokens","title":"KTAE: A Model-Free Algorithm to Key-Tokens Advantage Estimation in Mathematical Reasoning","date":"2025-05-22","arxiv_id":"2505.16826","n_code_links":1,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":0}},{"paper":"/paper/disco-reinforcing-large-reasoning-models-with","title":"DisCO: Reinforcing Large Reasoning Models with Discriminative Constrained Optimization","date":"2025-05-18","arxiv_id":"2505.12366","n_code_links":1,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":0}},{"paper":null,"title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","date":"2025-04-07","arxiv_id":"2504.05118","n_code_links":0,"syntology":null},{"paper":null,"title":"Improving Multi-Step Reasoning Abilities of Large Language Models with Direct Advantage Policy Optimization","date":"2024-12-24","arxiv_id":"2412.18279","n_code_links":0,"syntology":null},{"paper":null,"title":"Dual Approximation Policy Optimization","date":"2024-10-02","arxiv_id":"2410.01249","n_code_links":0,"syntology":null},{"paper":"/paper/task-specific-objectives-of-pre-trained","title":"Dialogue-adaptive Language Model Pre-training From Quality Estimation","date":"2020-09-10","arxiv_id":"2009.04984","n_code_links":1,"syntology":null}],"papers_shown":11,"tasks":[{"task":"/task/mathematical-reasoning","name":"Mathematical Reasoning","papers":2},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":2},{"task":null,"name":"8k","papers":1},{"task":"/task/gsm8k","name":"GSM8K","papers":1},{"task":"/task/informativeness","name":"Informativeness","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/multi-hop-question-answering","name":"Multi-hop Question Answering","papers":1},{"task":"/task/offline-rl","name":"Offline RL","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/scheduling","name":"Scheduling","papers":1},{"task":"/task/self-supervised-learning","name":"Self-Supervised Learning","papers":1},{"task":"/task/specificity","name":"Specificity","papers":1},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":1}],"tasks_shown":16,"n_tasks":16,"usage_by_year":[{"year":"2020","papers":1},{"year":"2024","papers":2},{"year":"2025","papers":8}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/dapo"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}