{"url":"/method/step-dpo","slug":"step-dpo","name":"Step-DPO","full_name":"Step-wise Direct Preference Optimization","full_name_withheld":false,"description_markdown":null,"description_state":"placeholder","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":null,"title":null,"url_on_a_paper_host":false},"code_snippet_url":"https://github.com/dvlab-research/Step-DPO","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":2,"archive_num_papers":2,"papers_newest_first":[{"paper":null,"title":"Full-Step-DPO: Self-Supervised Preference Optimization with Step-wise Rewards for Mathematical Reasoning","date":"2025-02-20","arxiv_id":"2502.14356","n_code_links":0,"syntology":null},{"paper":"/paper/step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","n_code_links":1,"syntology":{"ran":7,"of":12,"unverified":5,"pointer_only":12}}],"papers_shown":2,"tasks":[{"task":"/task/mathematical-reasoning","name":"Mathematical Reasoning","papers":2},{"task":"/task/arithmetic-reasoning","name":"Arithmetic Reasoning","papers":1},{"task":"/task/gsm8k","name":"GSM8K","papers":1},{"task":"/task/math","name":"Math","papers":1},{"task":"/task/math-word-problem-solving","name":"Math Word Problem Solving","papers":1}],"tasks_shown":5,"n_tasks":5,"usage_by_year":[{"year":"2024","papers":1},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/step-dpo"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}