{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dpo/papers/2","list_of":"/method/dpo","method":"DPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":409,"counts":{"archive_papers_tagged":409,"with_a_code_link":184,"where_syntology_ran_a_sample":112,"not_listed_spam_title":0,"listed":409,"listed_where_code_ran":112,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":98,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":98,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dpo","prev":"/method/dpo","next":"/method/dpo/papers/3","papers":[{"paper":"/paper/agentic-reward-modeling-integrating-human","slug":"agentic-reward-modeling-integrating-human","title":"Agentic Reward Modeling: Integrating Human Preferences with Verifiable Correctness Signals for Reliable Reward Systems","date":"2025-02-26","arxiv_id":"2502.19328","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":8,"n_instrument":4,"unverified":2,"pointer_only":0,"phrase":"12 ran (of which 3 constructed an object rather than computing a result; 8 with no instrument failure: 4 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-keg/agentic-reward-modeling"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":3,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-rlhf-be-more-efficient-with-imperfect","slug":"can-rlhf-be-more-efficient-with-imperfect","title":"Can RLHF be More Efficient with Imperfect Reward Models? A Policy Coverage Perspective","date":"2025-02-26","arxiv_id":"2502.19255","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jiaweihhuang/rlhf_rewardtransfer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"paper":null,"slug":"onerec-unifying-retrieve-and-rank-with","title":"OneRec: Unifying Retrieve and Rank with Generative Recommender and Iterative Preference Alignment","date":"2025-02-26","arxiv_id":"2502.18965","n_code_links":0,"syntology":null},{"paper":null,"slug":"debt-collection-negotiations-with-large","title":"Debt Collection Negotiations with Large Language Models: An Evaluation System and Optimizing Decision Making with Multi-Agent","date":"2025-02-25","arxiv_id":"2502.18228","n_code_links":0,"syntology":null},{"paper":null,"slug":"aligning-compound-ai-systems-via-system-level","title":"Aligning Compound AI Systems via System-level DPO","date":"2025-02-24","arxiv_id":"2502.17721","n_code_links":0,"syntology":null},{"paper":null,"slug":"finding-the-sweet-spot-preference-data","title":"Finding the Sweet Spot: Preference Data Construction for Scaling Preference Optimization","date":"2025-02-24","arxiv_id":"2502.16825","n_code_links":0,"syntology":null},{"paper":"/paper/hippo-enhancing-the-table-understanding","slug":"hippo-enhancing-the-table-understanding","title":"HIPPO: Enhancing the Table Understanding Capability of Large Language Models through Hybrid-Modal Preference Optimization","date":"2025-02-24","arxiv_id":"2502.17315","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["neuir/hippo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/codesync-synchronizing-large-language-models","slug":"codesync-synchronizing-large-language-models","title":"CODESYNC: Synchronizing Large Language Models with Dynamic Code Evolution at Scale","date":"2025-02-23","arxiv_id":"2502.16645","n_code_links":1,"syntology":null},{"paper":null,"slug":"c-3dpo-constrained-controlled-classification","title":"C-3DPO: Constrained Controlled Classification for Direct Preference Optimization","date":"2025-02-22","arxiv_id":"2502.17507","n_code_links":0,"syntology":null},{"paper":"/paper/earlier-tokens-contribute-more-learning","slug":"earlier-tokens-contribute-more-learning","title":"Earlier Tokens Contribute More: Learning Direct Preference Optimization From Temporal Decay Perspective","date":"2025-02-20","arxiv_id":"2502.14340","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lotusrc/d2po"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"federated-fine-tuning-of-large-language-1","title":"Federated Fine-Tuning of Large Language Models: Kahneman-Tversky vs. Direct Preference Optimization","date":"2025-02-20","arxiv_id":"2502.14187","n_code_links":0,"syntology":null},{"paper":null,"slug":"full-step-dpo-self-supervised-preference","title":"Full-Step-DPO: Self-Supervised Preference Optimization with Step-wise Rewards for Mathematical Reasoning","date":"2025-02-20","arxiv_id":"2502.14356","n_code_links":0,"syntology":null},{"paper":"/paper/length-controlled-margin-based-preference","slug":"length-controlled-margin-based-preference","title":"Length-Controlled Margin-Based Preference Optimization without Reference Model","date":"2025-02-20","arxiv_id":"2502.14643","n_code_links":1,"syntology":null},{"paper":null,"slug":"less-is-more-improving-llm-alignment-via","title":"Less is More: Improving LLM Alignment via Preference Data Selection","date":"2025-02-20","arxiv_id":"2502.14560","n_code_links":0,"syntology":null},{"paper":"/paper/self-improvement-towards-pareto-optimality","slug":"self-improvement-towards-pareto-optimality","title":"Self-Improvement Towards Pareto Optimality: Mitigating Preference Conflicts in Multi-Objective Alignment","date":"2025-02-20","arxiv_id":"2502.14354","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-safety-retrofitting-against","title":"Efficient Safety Retrofitting Against Jailbreaking for LLMs","date":"2025-02-19","arxiv_id":"2502.13603","n_code_links":0,"syntology":null},{"paper":"/paper/longpo-long-context-self-evolution-of-large","slug":"longpo-long-context-self-evolution-of-large","title":"LongPO: Long Context Self-Evolution of Large Language Models through Short-to-Long Preference Optimization","date":"2025-02-19","arxiv_id":"2502.13922","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DAMO-NLP-SG/LongPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/kl-penalty-control-via-perturbation-for","slug":"kl-penalty-control-via-perturbation-for","title":"KL Penalty Control via Perturbation for Direct Preference Optimization","date":"2025-02-18","arxiv_id":"2502.13177","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-step-alignment-as-markov-games-an","title":"Multi-Step Alignment as Markov Games: An Optimistic Online Gradient Descent Approach with Convergence Guarantees","date":"2025-02-18","arxiv_id":"2502.12678","n_code_links":0,"syntology":null},{"paper":null,"slug":"constructa-automating-commercial-construction","title":"CONSTRUCTA: Automating Commercial Construction Schedules in Fabrication Facilities with Large Language Models","date":"2025-02-17","arxiv_id":"2502.12066","n_code_links":0,"syntology":null},{"paper":null,"slug":"improve-llm-as-a-judge-ability-as-a-general","title":"Improve LLM-as-a-Judge Ability as a General Ability","date":"2025-02-17","arxiv_id":"2502.11689","n_code_links":0,"syntology":null},{"paper":"/paper/uncovering-the-impact-of-chain-of-thought","slug":"uncovering-the-impact-of-chain-of-thought","title":"Uncovering the Impact of Chain-of-Thought Reasoning for Direct Preference Optimization: Lessons from Text-to-SQL","date":"2025-02-17","arxiv_id":"2502.11656","n_code_links":1,"syntology":null},{"paper":null,"slug":"balancing-the-budget-understanding-trade-offs","title":"Balancing the Budget: Understanding Trade-offs Between Supervised and Preference-Based Finetuning","date":"2025-02-16","arxiv_id":"2502.11284","n_code_links":0,"syntology":null},{"paper":null,"slug":"preference-learning-made-easy-everything","title":"Preference learning made easy: Everything should be understood through win rate","date":"2025-02-14","arxiv_id":"2502.10505","n_code_links":0,"syntology":null},{"paper":"/paper/step-video-t2v-technical-report-the-practice","slug":"step-video-t2v-technical-report-the-practice","title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","date":"2025-02-14","arxiv_id":"2502.10248","n_code_links":3,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stepfun-ai/step-video-t2v"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/dpo-shift-shifting-the-distribution-of-direct","slug":"dpo-shift-shifting-the-distribution-of-direct","title":"DPO-Shift: Shifting the Distribution of Direct Preference Optimization","date":"2025-02-11","arxiv_id":"2502.07599","n_code_links":1,"syntology":null},{"paper":"/paper/principled-data-selection-for-alignment-the","slug":"principled-data-selection-for-alignment-the","title":"Principled Data Selection for Alignment: The Hidden Risks of Difficult Examples","date":"2025-02-11","arxiv_id":"2502.09650","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["glorgao/selectivedpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"design-considerations-in-offline-preference","title":"Design Considerations in Offline Preference-based RL","date":"2025-02-08","arxiv_id":"2502.06861","n_code_links":0,"syntology":null},{"paper":"/paper/llms-can-teach-themselves-to-better-predict","slug":"llms-can-teach-themselves-to-better-predict","title":"LLMs Can Teach Themselves to Better Predict the Future","date":"2025-02-07","arxiv_id":"2502.05253","n_code_links":1,"syntology":null},{"paper":null,"slug":"direct-distributional-optimization-for","title":"Direct Distributional Optimization for Provable Alignment of Diffusion Models","date":"2025-02-05","arxiv_id":"2502.02954","n_code_links":0,"syntology":null},{"paper":"/paper/damo-data-and-model-aware-alignment-of-multi","slug":"damo-data-and-model-aware-alignment-of-multi","title":"DAMO: Data- and Model-aware Alignment of Multi-modal LLMs","date":"2025-02-04","arxiv_id":"2502.01943","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributionally-robust-direct-preference","title":"Distributionally Robust Direct Preference Optimization","date":"2025-02-04","arxiv_id":"2502.01930","n_code_links":0,"syntology":null},{"paper":null,"slug":"harness-local-rewards-for-global-benefits","title":"Harness Local Rewards for Global Benefits: Effective Text-to-Video Generation Alignment with Patch-level Reward Models","date":"2025-02-04","arxiv_id":"2502.06812","n_code_links":0,"syntology":null},{"paper":"/paper/longdpo-unlock-better-long-form-generation","slug":"longdpo-unlock-better-long-form-generation","title":"LongDPO: Unlock Better Long-form Generation Abilities for LLMs via Critique-augmented Stepwise Information","date":"2025-02-04","arxiv_id":"2502.02095","n_code_links":1,"syntology":null},{"paper":null,"slug":"eliciting-language-model-behaviors-with","title":"Eliciting Language Model Behaviors with Investigator Agents","date":"2025-02-03","arxiv_id":"2502.01236","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-differences-between-direct-alignment","title":"The Differences Between Direct Alignment Algorithms are a Blur","date":"2025-02-03","arxiv_id":"2502.01237","n_code_links":0,"syntology":null},{"paper":null,"slug":"huvidpo-enhancing-video-generation-through","title":"HuViDPO:Enhancing Video Generation through Direct Preference Optimization for Human-Centric Alignment","date":"2025-02-02","arxiv_id":"2502.01690","n_code_links":0,"syntology":null},{"paper":null,"slug":"refining-alignment-framework-for-diffusion","title":"Refining Alignment Framework for Diffusion Models with Intermediate-Step Preference Ranking","date":"2025-02-01","arxiv_id":"2502.01667","n_code_links":0,"syntology":null},{"paper":"/paper/guardreasoner-towards-reasoning-based-llm","slug":"guardreasoner-towards-reasoning-based-llm","title":"GuardReasoner: Towards Reasoning-based LLM Safeguards","date":"2025-01-30","arxiv_id":"2501.18492","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yueliu1999/guardreasoner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/wildchat-50m-a-deep-dive-into-the-role-of","slug":"wildchat-50m-a-deep-dive-into-the-role-of","title":"WILDCHAT-50M: A Deep Dive Into the Role of Synthetic Data in Post-Training","date":"2025-01-30","arxiv_id":"2501.18511","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["penfever/wildchat-50m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/chip-cross-modal-hierarchical-direct","slug":"chip-cross-modal-hierarchical-direct","title":"CHiP: Cross-modal Hierarchical Direct Preference Optimization for Multimodal LLMs","date":"2025-01-28","arxiv_id":"2501.16629","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":1,"n_instrument":4,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lvugai/chip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mitigating-hallucinations-in-large-vision-3","slug":"mitigating-hallucinations-in-large-vision-3","title":"Mitigating Hallucinations in Large Vision-Language Models via DPO: On-Policy Data Hold the Key","date":"2025-01-16","arxiv_id":"2501.09695","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":9,"n_instrument":2,"unverified":5,"pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["zhyang2226/opa-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamic-portfolio-optimization-via-augmented","title":"Dynamic Portfolio Optimization via Augmented DDPG with Quantum Price Levels-Based Trading Strategy","date":"2025-01-15","arxiv_id":"2501.08528","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-label-refinement-matters-more-than","slug":"iterative-label-refinement-matters-more-than","title":"Iterative Label Refinement Matters More than Preference Optimization under Weak Supervision","date":"2025-01-14","arxiv_id":"2501.07886","n_code_links":1,"syntology":null},{"paper":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"focalpo-enhancing-preference-optimizing-by","title":"FocalPO: Enhancing Preference Optimizing by Focusing on Correct Preference Rankings","date":"2025-01-11","arxiv_id":"2501.06645","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-preference-fine-tuning-of","title":"Personalized Preference Fine-tuning of Diffusion Models","date":"2025-01-11","arxiv_id":"2501.06655","n_code_links":0,"syntology":null},{"paper":null,"slug":"many-of-your-dpos-are-secretly-one-attempting","title":"Many of Your DPOs are Secretly One: Attempting Unification Through Mutual Information","date":"2025-01-02","arxiv_id":"2501.01544","n_code_links":0,"syntology":null},{"paper":"/paper/rrhf-v-ranking-responses-to-mitigate","slug":"rrhf-v-ranking-responses-to-mitigate","title":"RRHF-V: Ranking Responses to Mitigate Hallucinations in Multimodal Large Language Models with Human Feedback","date":"2025-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"plug-and-play-training-framework-for","title":"Plug-and-Play Training Framework for Preference Optimization","date":"2024-12-30","arxiv_id":"2412.20996","n_code_links":0,"syntology":null},{"paper":"/paper/no-preference-left-behind-group","slug":"no-preference-left-behind-group","title":"No Preference Left Behind: Group Distributional Preference Optimization","date":"2024-12-28","arxiv_id":"2412.20299","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["BigBinnie/GDPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multimodal-preference-data-synthetic","slug":"multimodal-preference-data-synthetic","title":"Multimodal Preference Data Synthetic Alignment with Reward Model","date":"2024-12-23","arxiv_id":"2412.17417","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-the-logic-of-direct-preference","title":"Understanding the Logic of Direct Preference Alignment through Logic","date":"2024-12-23","arxiv_id":"2412.17696","n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-llms-to-refine-with-tools","title":"Teaching LLMs to Refine with Tools","date":"2024-12-22","arxiv_id":"2412.16871","n_code_links":0,"syntology":null},{"paper":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"northeastern-uni-at-multilingual","title":"Northeastern Uni at Multilingual Counterspeech Generation: Enhancing Counter Speech Generation with LLM Alignment through Direct Preference Optimization","date":"2024-12-19","arxiv_id":"2412.15453","n_code_links":0,"syntology":null},{"paper":null,"slug":"onlinevpo-align-video-diffusion-model-with","title":"OnlineVPO: Align Video Diffusion Model with Online Video-Centric Preference Optimization","date":"2024-12-19","arxiv_id":"2412.15159","n_code_links":0,"syntology":null},{"paper":null,"slug":"energy-based-preference-model-offers-better","title":"Energy-Based Preference Model Offers Better Offline Alignment than the Bradley-Terry Preference Model","date":"2024-12-18","arxiv_id":"2412.13862","n_code_links":0,"syntology":null},{"paper":"/paper/preference-oriented-supervised-fine-tuning","slug":"preference-oriented-supervised-fine-tuning","title":"Preference-Oriented Supervised Fine-Tuning: Favoring Target Model Over Aligned Large Language Models","date":"2024-12-17","arxiv_id":"2412.12865","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Savannah120/alignment-handbook-PoFT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"safetydpo-scalable-safety-alignment-for-text","title":"SafetyDPO: Scalable Safety Alignment for Text-to-Image Generation","date":"2024-12-13","arxiv_id":"2412.10493","n_code_links":0,"syntology":null},{"paper":null,"slug":"sail-into-the-headwind-alignment-via-robust","title":"Sail into the Headwind: Alignment via Robust Rewards and Dynamic Labels against Reward Hacking","date":"2024-12-12","arxiv_id":"2412.09544","n_code_links":0,"syntology":null},{"paper":"/paper/sprec-leveraging-self-play-to-debias","slug":"sprec-leveraging-self-play-to-debias","title":"SPRec: Leveraging Self-Play to Debias Preference Alignment for Large Language Model-based Recommendations","date":"2024-12-12","arxiv_id":"2412.09243","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["regionch/sprec"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"knowledge-graph-guided-evaluation-of","title":"Knowledge Graph Guided Evaluation of Abstention Techniques","date":"2024-12-10","arxiv_id":"2412.07430","n_code_links":0,"syntology":null},{"paper":null,"slug":"silmm-self-improving-large-multimodal-models","title":"SILMM: Self-Improving Large Multimodal Models for Compositional Text-to-Image Generation","date":"2024-12-08","arxiv_id":"2412.05818","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-ingredient","slug":"large-language-models-for-ingredient","title":"Large Language Models for Ingredient Substitution in Food Recipes using Supervised Fine-tuning and Direct Preference Optimization","date":"2024-12-06","arxiv_id":"2412.04922","n_code_links":1,"syntology":null},{"paper":"/paper/sopo-text-to-motion-generation-using-semi","slug":"sopo-text-to-motion-generation-using-semi","title":"SoPo: Text-to-Motion Generation Using Semi-Online Preference Optimization","date":"2024-12-06","arxiv_id":"2412.05095","n_code_links":1,"syntology":null},{"paper":"/paper/patchdpo-patch-level-dpo-for-finetuning-free","slug":"patchdpo-patch-level-dpo-for-finetuning-free","title":"PatchDPO: Patch-level DPO for Finetuning-free Personalized Image Generation","date":"2024-12-04","arxiv_id":"2412.03177","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":8,"n_instrument":2,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hqhqaq/patchdpo"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mars-po-multi-agent-reasoning-system","title":"Mars-PO: Multi-Agent Reasoning System Preference Optimization","date":"2024-11-28","arxiv_id":"2411.19039","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-the-transferability-of-adversarial-4","title":"Improving the Transferability of Adversarial Attacks on Face Recognition with Diverse Parameters Augmentation","date":"2024-11-23","arxiv_id":"2411.15555","n_code_links":0,"syntology":null},{"paper":null,"slug":"reward-fine-tuning-two-step-diffusion-models","title":"Reward Fine-Tuning Two-Step Diffusion Models via Learning Differentiable Latent-Space Surrogate Reward","date":"2024-11-22","arxiv_id":"2411.15247","n_code_links":0,"syntology":null},{"paper":"/paper/insight-v-exploring-long-chain-visual","slug":"insight-v-exploring-long-chain-visual","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","date":"2024-11-21","arxiv_id":"2411.14432","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":0,"n_instrument":5,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["dongyh20/insight-v"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"entropy-controllable-direct-preference","title":"Entropy Controllable Direct Preference Optimization","date":"2024-11-12","arxiv_id":"2411.07595","n_code_links":0,"syntology":null},{"paper":"/paper/ablation-is-not-enough-to-emulate-dpo-how","slug":"ablation-is-not-enough-to-emulate-dpo-how","title":"Beyond Toxic Neurons: A Mechanistic Analysis of DPO for Toxicity Reduction","date":"2024-11-10","arxiv_id":"2411.06424","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yushi-y/dpo-toxic-neurons"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-loss-landscapes-in-preference","title":"Learning Loss Landscapes in Preference Optimization","date":"2024-11-10","arxiv_id":"2411.06568","n_code_links":0,"syntology":null},{"paper":"/paper/iopo-empowering-llms-with-complex-instruction","slug":"iopo-empowering-llms-with-complex-instruction","title":"IOPO: Empowering LLMs with Complex Instruction Following via Input-Output Preference Optimization","date":"2024-11-09","arxiv_id":"2411.06208","n_code_links":1,"syntology":null},{"paper":"/paper/abstract2appendix-academic-reviews-enhance","slug":"abstract2appendix-academic-reviews-enhance","title":"Abstract2Appendix: Academic Reviews Enhance LLM Long-Context Capabilities","date":"2024-11-07","arxiv_id":"2411.05232","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-improved-preference-optimization","title":"Towards Improved Preference Optimization Pipeline: from Data Generation to Budget-Controlled Regularization","date":"2024-11-07","arxiv_id":"2411.05875","n_code_links":0,"syntology":null},{"paper":null,"slug":"see-dpo-self-entropy-enhanced-direct","title":"SEE-DPO: Self Entropy Enhanced Direct Preference Optimization","date":"2024-11-06","arxiv_id":"2411.04712","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-root-shapes-the-fruit-on-the-persistence","title":"The Root Shapes the Fruit: On the Persistence of Gender-Exclusive Harms in Aligned Language Models","date":"2024-11-06","arxiv_id":"2411.03700","n_code_links":0,"syntology":null},{"paper":"/paper/sample-efficient-alignment-for-llms","slug":"sample-efficient-alignment-for-llms","title":"Sample-Efficient Alignment for LLMs","date":"2024-11-03","arxiv_id":"2411.01493","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/oat"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/todo-enhancing-llm-alignment-with-ternary","slug":"todo-enhancing-llm-alignment-with-ternary","title":"TODO: Enhancing LLM Alignment with Ternary Preferences","date":"2024-11-02","arxiv_id":"2411.02442","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["xxares/todo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evolving-alignment-via-asymmetric-self-play","title":"Scalable Reinforcement Post-Training Beyond Static Human Prompts: Evolving Alignment via Asymmetric Self-Play","date":"2024-10-31","arxiv_id":"2411.00062","n_code_links":0,"syntology":null},{"paper":"/paper/vpo-leveraging-the-number-of-votes-in","slug":"vpo-leveraging-the-number-of-votes-in","title":"VPO: Leveraging the Number of Votes in Preference Optimization","date":"2024-10-30","arxiv_id":"2410.22891","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ku-dmlab/vpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/f-po-generalizing-preference-optimization","slug":"f-po-generalizing-preference-optimization","title":"$f$-PO: Generalizing Preference Optimization with $f$-divergence Minimization","date":"2024-10-29","arxiv_id":"2410.21662","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["minkaixu/fpo"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"flow-dpo-improving-llm-mathematical-reasoning","title":"Flow-DPO: Improving LLM Mathematical Reasoning through Online Multi-Agent Learning","date":"2024-10-29","arxiv_id":"2410.22304","n_code_links":0,"syntology":null},{"paper":"/paper/longreward-improving-long-context-large","slug":"longreward-improving-long-context-large","title":"LongReward: Improving Long-context Large Language Models with AI Feedback","date":"2024-10-28","arxiv_id":"2410.21252","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THUDM/LongReward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/accelerating-direct-preference-optimization","slug":"accelerating-direct-preference-optimization","title":"Accelerating Direct Preference Optimization with Prefix Sharing","date":"2024-10-27","arxiv_id":"2410.20305","n_code_links":1,"syntology":null},{"paper":"/paper/learning-from-response-not-preference-a","slug":"learning-from-response-not-preference-a","title":"Learning from Response not Preference: A Stackelberg Approach for LLM Detoxification using Non-parallel Data","date":"2024-10-27","arxiv_id":"2410.20298","n_code_links":1,"syntology":null},{"paper":"/paper/fast-best-of-n-decoding-via-speculative","slug":"fast-best-of-n-decoding-via-speculative","title":"Fast Best-of-N Decoding via Speculative Rejection","date":"2024-10-26","arxiv_id":"2410.20290","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Zanette-Labs/SpeculativeRejection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"uncertainty-penalized-direct-preference","title":"Uncertainty-Penalized Direct Preference Optimization","date":"2024-10-26","arxiv_id":"2410.20187","n_code_links":0,"syntology":null},{"paper":null,"slug":"2d-dpo-scaling-direct-preference-optimization","title":"2D-DPO: Scaling Direct Preference Optimization with 2-Dimensional Supervision","date":"2024-10-25","arxiv_id":"2410.19720","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-inverse-folding-for-peptide-design","title":"Improving Inverse Folding for Peptide Design with Diversity-regularized Direct Preference Optimization","date":"2024-10-25","arxiv_id":"2410.19471","n_code_links":0,"syntology":null},{"paper":null,"slug":"aligning-codellms-with-direct-preference","title":"Aligning CodeLLMs with Direct Preference Optimization","date":"2024-10-24","arxiv_id":"2410.18585","n_code_links":0,"syntology":null},{"paper":"/paper/asynchronous-rlhf-faster-and-more-efficient","slug":"asynchronous-rlhf-faster-and-more-efficient","title":"Asynchronous RLHF: Faster and More Efficient Off-Policy RL for Language Models","date":"2024-10-23","arxiv_id":"2410.18252","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["mnoukhov/async_rlhf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"scalable-ranked-preference-optimization-for","title":"Scalable Ranked Preference Optimization for Text-to-Image Generation","date":"2024-10-23","arxiv_id":"2410.18013","n_code_links":0,"syntology":null},{"paper":null,"slug":"advweb-controllable-black-box-attacks-on-vlm","title":"AdvAgent: Controllable Blackbox Red-teaming on Web Agents","date":"2024-10-22","arxiv_id":"2410.17401","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-llms-with-direct-preferences-a","title":"Optimizing LLMs with Direct Preferences: A Data Efficiency Perspective","date":"2024-10-22","arxiv_id":"2410.16586","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comprehensive-survey-of-datasets-theories","title":"A Comprehensive Survey of Direct Preference Optimization: Datasets, Theories, Variants, and Applications","date":"2024-10-21","arxiv_id":"2410.15595","n_code_links":0,"syntology":null},{"paper":null,"slug":"gdpo-learning-to-directly-align-language","title":"GDPO: Learning to Directly Align Language Models with Diversity Using GFlowNets","date":"2024-10-19","arxiv_id":"2410.15096","n_code_links":0,"syntology":null}],"record_sha256":"eb02953cb89f0f61cb4608ccdc2cdd07609557beef77c784b868ff22dd2ad6ad","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}