{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dpo/papers/4","list_of":"/method/dpo","method":"DPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":409,"counts":{"archive_papers_tagged":409,"with_a_code_link":184,"where_syntology_ran_a_sample":112,"not_listed_spam_title":0,"listed":409,"listed_where_code_ran":112,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":98,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":98,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dpo","prev":"/method/dpo/papers/3","next":"/method/dpo/papers/5","papers":[{"paper":null,"slug":"online-dpo-online-direct-preference","title":"Online DPO: Online Direct Preference Optimization with Fast-Slow Chasing","date":"2024-06-08","arxiv_id":"2406.05534","n_code_links":0,"syntology":null},{"paper":"/paper/step-aware-preference-optimization-aligning","slug":"step-aware-preference-optimization-aligning","title":"Aesthetic Post-Training Diffusion Models from Generic Preferences with Step-by-step Preference Optimization","date":"2024-06-06","arxiv_id":"2406.04314","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["rockeycoss/spo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"is-free-self-alignment-possible","title":"Is Free Self-Alignment Possible?","date":"2024-06-05","arxiv_id":"2406.03642","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-beyond-english-scaling-the-multilingual","title":"LLMs Beyond English: Scaling the Multilingual Capability of LLMs with Cross-Lingual Feedback","date":"2024-06-03","arxiv_id":"2406.01771","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-improving-robust-preference-optimization","title":"Self-Improving Robust Preference Optimization","date":"2024-06-03","arxiv_id":"2406.01660","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-preference-fine-tuning-through","title":"The Importance of Online Data: Understanding Preference Fine-tuning via Coverage","date":"2024-06-03","arxiv_id":"2406.01462","n_code_links":0,"syntology":null},{"paper":null,"slug":"direct-alignment-of-language-models-via","title":"Direct Alignment of Language Models via Quality-Aware Self-Refinement","date":"2024-05-31","arxiv_id":"2405.21040","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploratory-preference-optimization","title":"Exploratory Preference Optimization: Harnessing Implicit Q*-Approximation for Sample-Efficient RLHF","date":"2024-05-31","arxiv_id":"2405.21046","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-clarify-multi-turn-conversations","title":"Learning to Clarify: Multi-turn Conversations with Action-Based Contrastive Self-Training","date":"2024-05-31","arxiv_id":"2406.00222","n_code_links":0,"syntology":null},{"paper":"/paper/self-augmented-preference-optimization-off","slug":"self-augmented-preference-optimization-off","title":"Self-Augmented Preference Optimization: Off-Policy Paradigms for Language Model Alignment","date":"2024-05-31","arxiv_id":"2405.20830","n_code_links":1,"syntology":null},{"paper":null,"slug":"boost-your-own-human-image-generation-model","title":"Boost Your Human Image Generation Model via Direct Preference Optimization","date":"2024-05-30","arxiv_id":"2405.20216","n_code_links":0,"syntology":null},{"paper":"/paper/xwin-lm-strong-and-scalable-alignment","slug":"xwin-lm-strong-and-scalable-alignment","title":"Xwin-LM: Strong and Scalable Alignment Practice for LLMs","date":"2024-05-30","arxiv_id":"2405.20335","n_code_links":1,"syntology":null},{"paper":null,"slug":"preference-learning-algorithms-do-not-learn","title":"Preference Learning Algorithms Do Not Learn Preference Rankings","date":"2024-05-29","arxiv_id":"2405.19534","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-preference-optimization-through-reward","title":"Robust Preference Optimization through Reward Model Distillation","date":"2024-05-29","arxiv_id":"2405.19316","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-preference-optimization-augmenting","title":"Unified Preference Optimization: Language Model Alignment Beyond the Preference Frontier","date":"2024-05-28","arxiv_id":"2405.17956","n_code_links":0,"syntology":null},{"paper":"/paper/online-merging-optimizers-for-boosting","slug":"online-merging-optimizers-for-boosting","title":"Online Merging Optimizers for Boosting Rewards and Mitigating Tax in Alignment","date":"2024-05-28","arxiv_id":"2405.17931","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-reference-preference-optimization-for","title":"Multi-Reference Preference Optimization for Large Language Models","date":"2024-05-26","arxiv_id":"2405.16388","n_code_links":0,"syntology":null},{"paper":null,"slug":"provably-mitigating-overoptimization-in-rlhf","title":"Provably Mitigating Overoptimization in RLHF: Your SFT Loss is Implicitly an Adversarial Regularizer","date":"2024-05-26","arxiv_id":"2405.16436","n_code_links":0,"syntology":null},{"paper":"/paper/triple-preference-optimization-achieving","slug":"triple-preference-optimization-achieving","title":"Triple Preference Optimization: Achieving Better Alignment with Less Data in a Single Step Optimization","date":"2024-05-26","arxiv_id":"2405.16681","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sahsaeedi/triple-preference-optimization"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"direct-preference-optimization-with","title":"Direct Preference Optimization With Unobserved Preference Heterogeneity","date":"2024-05-23","arxiv_id":"2405.15065","n_code_links":0,"syntology":null},{"paper":null,"slug":"mallows-dpo-fine-tune-your-llm-with","title":"MallowsPO: Fine-Tune Your LLM with Preference Dispersions","date":"2024-05-23","arxiv_id":"2405.14953","n_code_links":0,"syntology":null},{"paper":"/paper/simpo-simple-preference-optimization-with-a","slug":"simpo-simple-preference-optimization-with-a","title":"SimPO: Simple Preference Optimization with a Reference-Free Reward","date":"2024-05-23","arxiv_id":"2405.14734","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princeton-nlp/simpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/annotation-efficient-preference-optimization","slug":"annotation-efficient-preference-optimization","title":"Annotation-Efficient Preference Optimization for Language Model Alignment","date":"2024-05-22","arxiv_id":"2405.13541","n_code_links":1,"syntology":null},{"paper":"/paper/curriculum-direct-preference-optimization-for","slug":"curriculum-direct-preference-optimization-for","title":"Curriculum Direct Preference Optimization for Diffusion and Consistency Models","date":"2024-05-22","arxiv_id":"2405.13637","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":2,"n_instrument":1,"unverified":5,"pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["croitorualin/curriculum-dpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/detox-toxic-subspace-projection-for-model","slug":"detox-toxic-subspace-projection-for-model","title":"Model Editing as a Robust and Denoised variant of DPO: A Case Study on Toxicity","date":"2024-05-22","arxiv_id":"2405.13967","n_code_links":2,"syntology":{"ran":7,"of":12,"n_ran_checked":6,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["uppaal/detox-edit"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adversarial-dpo-harnessing-harmful-data-for","title":"Adversarial DPO: Harnessing Harmful Data for Reducing Toxicity with Minimal Impact on Coherence and Evasiveness in Dialogue Agents","date":"2024-05-21","arxiv_id":"2405.12900","n_code_links":0,"syntology":null},{"paper":"/paper/energy-rank-alignment-using-preference","slug":"energy-rank-alignment-using-preference","title":"Aligning Transformers with Continuous Feedback via Energy Rank Alignment","date":"2024-05-21","arxiv_id":"2405.12961","n_code_links":2,"syntology":null},{"paper":null,"slug":"mining-the-explainability-and-generalization","title":"Mining the Explainability and Generalization: Fact Verification Based on Self-Instruction","date":"2024-05-21","arxiv_id":"2405.12579","n_code_links":0,"syntology":null},{"paper":"/paper/openrlhf-an-easy-to-use-scalable-and-high","slug":"openrlhf-an-easy-to-use-scalable-and-high","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","date":"2024-05-20","arxiv_id":"2405.11143","n_code_links":4,"syntology":null},{"paper":"/paper/quantifying-and-optimizing-global","slug":"quantifying-and-optimizing-global","title":"Quantifying and Optimizing Global Faithfulness in Persona-driven Role-playing","date":"2024-05-13","arxiv_id":"2405.07726","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["KomeijiForce/Active_Passive_Constraint_Koishiday_2024"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"advanced-natural-based-interaction-for-the","title":"Advanced Natural-based interaction for the ITAlian language: LLaMAntino-3-ANITA","date":"2024-05-11","arxiv_id":"2405.07101","n_code_links":0,"syntology":null},{"paper":"/paper/value-augmented-sampling-for-language-model","slug":"value-augmented-sampling-for-language-model","title":"Value Augmented Sampling for Language Model Alignment and Personalization","date":"2024-05-10","arxiv_id":"2405.06639","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["idanshen/Value-Augmented-Sampling"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"modipo-text-to-motion-alignment-via-ai","title":"MoDiPO: text-to-motion alignment via AI-feedback-driven Direct Preference Optimization","date":"2024-05-06","arxiv_id":"2405.03803","n_code_links":0,"syntology":null},{"paper":"/paper/d2po-discriminator-guided-dpo-with-response","slug":"d2po-discriminator-guided-dpo-with-response","title":"D2PO: Discriminator-Guided DPO with Response Evaluation Models","date":"2024-05-02","arxiv_id":"2405.01511","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["PrasannS/d2po"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-play-preference-optimization-for","slug":"self-play-preference-optimization-for","title":"Self-Play Preference Optimization for Language Model Alignment","date":"2024-05-01","arxiv_id":"2405.00675","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["uclaml/sppo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"iterative-reasoning-preference-optimization","title":"Iterative Reasoning Preference Optimization","date":"2024-04-30","arxiv_id":"2404.19733","n_code_links":0,"syntology":null},{"paper":"/paper/dpo-meets-ppo-reinforced-token-optimization","slug":"dpo-meets-ppo-reinforced-token-optimization","title":"DPO Meets PPO: Reinforced Token Optimization for RLHF","date":"2024-04-29","arxiv_id":"2404.18922","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zkshan2002/rto"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","n_code_links":3,"syntology":{"ran":16,"of":20,"n_ran_checked":12,"n_instrument":4,"unverified":4,"pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/weak-to-strong-extrapolation-expedites","slug":"weak-to-strong-extrapolation-expedites","title":"Weak-to-Strong Extrapolation Expedites Alignment","date":"2024-04-25","arxiv_id":"2404.16792","n_code_links":1,"syntology":null},{"paper":"/paper/dpo-differential-reinforcement-learning-with","slug":"dpo-differential-reinforcement-learning-with","title":"DPO: A Differential and Pointwise Control Approach to Reinforcement Learning","date":"2024-04-24","arxiv_id":"2404.15617","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/insights-into-alignment-evaluating-dpo-and","slug":"insights-into-alignment-evaluating-dpo-and","title":"Insights into Alignment: Evaluating DPO and its Variants Across Multiple Tasks","date":"2024-04-23","arxiv_id":"2404.14723","n_code_links":1,"syntology":null},{"paper":"/paper/filtered-direct-preference-optimization","slug":"filtered-direct-preference-optimization","title":"Filtered Direct Preference Optimization","date":"2024-04-22","arxiv_id":"2404.13846","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["cyberagentailab/filtered-dpo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"from-r-to-q-your-language-model-is-secretly-a","title":"From $r$ to $Q^*$: Your Language Model is Secretly a Q-Function","date":"2024-04-18","arxiv_id":"2404.12358","n_code_links":0,"syntology":null},{"paper":"/paper/openbezoar-small-cost-effective-and-open","slug":"openbezoar-small-cost-effective-and-open","title":"OpenBezoar: Small, Cost-Effective and Open Models Trained on Mixes of Instruction Data","date":"2024-04-18","arxiv_id":"2404.12195","n_code_links":1,"syntology":null},{"paper":"/paper/token-level-direct-preference-optimization","slug":"token-level-direct-preference-optimization","title":"Token-level Direct Preference Optimization","date":"2024-04-18","arxiv_id":"2404.11999","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vance0124/token-level-direct-preference-optimization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/is-dpo-superior-to-ppo-for-llm-alignment-a","slug":"is-dpo-superior-to-ppo-for-llm-alignment-a","title":"Is DPO Superior to PPO for LLM Alignment? A Comprehensive Study","date":"2024-04-16","arxiv_id":"2404.10719","n_code_links":1,"syntology":null},{"paper":null,"slug":"latent-distance-guided-alignment-training-for","title":"Latent Distance Guided Alignment Training for Large Language Models","date":"2024-04-09","arxiv_id":"2404.06390","n_code_links":0,"syntology":null},{"paper":null,"slug":"binary-classifier-optimization-for-large","title":"Binary Classifier Optimization for Large Language Model Alignment","date":"2024-04-06","arxiv_id":"2404.04656","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-analyzing-and-understanding-the","title":"Towards Analyzing and Understanding the Limitations of DPO: A Theoretical Perspective","date":"2024-04-06","arxiv_id":"2404.04626","n_code_links":0,"syntology":null},{"paper":"/paper/direct-preference-optimization-of-video-large","slug":"direct-preference-optimization-of-video-large","title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","date":"2024-04-01","arxiv_id":"2404.01258","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["riflezhang/llava-hound-dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/regularized-best-of-n-sampling-to-mitigate","slug":"regularized-best-of-n-sampling-to-mitigate","title":"Regularized Best-of-N Sampling with Minimum Bayes Risk Objective for Language Model Alignment","date":"2024-04-01","arxiv_id":"2404.01054","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["CyberAgentAILab/regularized-bon"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/comparing-bad-apples-to-good-oranges-aligning","slug":"comparing-bad-apples-to-good-oranges-aligning","title":"Comparing Bad Apples to Good Oranges: Aligning Large Language Models via Joint Preference Optimization","date":"2024-03-31","arxiv_id":"2404.00530","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hritikbansal/dove"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/extensive-self-contrast-enables-feedback-free","slug":"extensive-self-contrast-enables-feedback-free","title":"Extensive Self-Contrast Enables Feedback-Free Language Model Alignment","date":"2024-03-31","arxiv_id":"2404.00604","n_code_links":2,"syntology":null},{"paper":"/paper/configurable-safety-tuning-of-language-models","slug":"configurable-safety-tuning-of-language-models","title":"Configurable Safety Tuning of Language Models with Synthetic Preference Data","date":"2024-03-30","arxiv_id":"2404.00495","n_code_links":1,"syntology":null},{"paper":null,"slug":"dialectical-alignment-resolving-the-tension","title":"Dialectical Alignment: Resolving the Tension of 3H and Security Threats of LLMs","date":"2024-03-30","arxiv_id":"2404.00486","n_code_links":0,"syntology":null},{"paper":"/paper/disentangling-length-from-quality-in-direct","slug":"disentangling-length-from-quality-in-direct","title":"Disentangling Length from Quality in Direct Preference Optimization","date":"2024-03-28","arxiv_id":"2403.19159","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"mixed-preference-optimization-reinforcement","title":"Mixed Preference Optimization: Reinforcement Learning with Data Selection and Better Reference Model","date":"2024-03-28","arxiv_id":"2403.19443","n_code_links":0,"syntology":null},{"paper":null,"slug":"sdpo-don-t-use-your-data-all-at-once","title":"sDPO: Don't Use Your Data All at Once","date":"2024-03-28","arxiv_id":"2403.19270","n_code_links":0,"syntology":null},{"paper":"/paper/detoxifying-large-language-models-via","slug":"detoxifying-large-language-models-via","title":"Detoxifying Large Language Models via Knowledge Editing","date":"2024-03-21","arxiv_id":"2403.14472","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"human-alignment-of-large-language-models","title":"Human Alignment of Large Language Models through Online Preference Optimisation","date":"2024-03-13","arxiv_id":"2403.08635","n_code_links":0,"syntology":null},{"paper":null,"slug":"curry-dpo-enhancing-alignment-using","title":"Curry-DPO: Enhancing Alignment using Curriculum Learning & Ranked Preferences","date":"2024-03-12","arxiv_id":"2403.07230","n_code_links":0,"syntology":null},{"paper":"/paper/negating-negatives-alignment-without-human","slug":"negating-negatives-alignment-without-human","title":"Negating Negatives: Alignment with Human Negative Samples via Distributional Dispreference Optimization","date":"2024-03-06","arxiv_id":"2403.03419","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-llm-safety-via-constrained-direct","title":"Enhancing LLM Safety via Constrained Direct Preference Optimization","date":"2024-03-04","arxiv_id":"2403.02475","n_code_links":0,"syntology":null},{"paper":null,"slug":"reward-model-learning-vs-direct-policy","title":"Reward Model Learning vs. Direct Policy Optimization: A Comparative Analysis of Learning from Human Preferences","date":"2024-03-04","arxiv_id":"2403.01857","n_code_links":0,"syntology":null},{"paper":"/paper/trial-and-error-exploration-based-trajectory","slug":"trial-and-error-exploration-based-trajectory","title":"Trial and Error: Exploration-Based Trajectory Optimization for LLM Agents","date":"2024-03-04","arxiv_id":"2403.02502","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yifan-song793/eto"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"provably-robust-dpo-aligning-language-models","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","date":"2024-03-01","arxiv_id":"2403.00409","n_code_links":0,"syntology":null},{"paper":null,"slug":"back-to-basics-revisiting-reinforce-style","title":"Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs","date":"2024-02-22","arxiv_id":"2402.14740","n_code_links":0,"syntology":null},{"paper":"/paper/smaug-fixing-failure-modes-of-preference","slug":"smaug-fixing-failure-modes-of-preference","title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","date":"2024-02-20","arxiv_id":"2402.13228","n_code_links":2,"syntology":null},{"paper":"/paper/direct-large-language-model-alignment-through","slug":"direct-large-language-model-alignment-through","title":"Direct Large Language Model Alignment Through Self-Rewarding Contrastive Prompt Distillation","date":"2024-02-19","arxiv_id":"2402.11907","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["exlaw/dlma"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"note-notable-generation-of-patient-text","title":"NOTE: Notable generation Of patient Text summaries through Efficient approach based on direct preference optimization","date":"2024-02-19","arxiv_id":"2402.11882","n_code_links":0,"syntology":null},{"paper":"/paper/direct-preference-optimization-with-an-offset","slug":"direct-preference-optimization-with-an-offset","title":"Direct Preference Optimization with an Offset","date":"2024-02-16","arxiv_id":"2402.10571","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rycolab/odpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-modal-preference-alignment-remedies","slug":"multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","arxiv_id":"2402.10884","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["findalexli/mllm-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"rs-dpo-a-hybrid-rejection-sampling-and-direct","title":"RS-DPO: A Hybrid Rejection Sampling and Direct Preference Optimization Method for Alignment of Large Language Models","date":"2024-02-15","arxiv_id":"2402.10038","n_code_links":0,"syntology":null},{"paper":"/paper/icdpo-effectively-borrowing-alignment","slug":"icdpo-effectively-borrowing-alignment","title":"ICDPO: Effectively Borrowing Alignment Capability of Others via In-context Direct Preference Optimization","date":"2024-02-14","arxiv_id":"2402.09320","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-from-human-feedback","title":"Reinforcement Learning from Human Feedback with Active Queries","date":"2024-02-14","arxiv_id":"2402.09401","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-preference-learning-for-large-language","title":"Active Preference Learning for Large Language Models","date":"2024-02-12","arxiv_id":"2402.08114","n_code_links":0,"syntology":null},{"paper":"/paper/refined-direct-preference-optimization-with","slug":"refined-direct-preference-optimization-with","title":"Refined Direct Preference Optimization with Synthetic Data for Behavioral Alignment of LLMs","date":"2024-02-12","arxiv_id":"2402.08005","n_code_links":1,"syntology":null},{"paper":"/paper/relative-preference-optimization-enhancing","slug":"relative-preference-optimization-enhancing","title":"Relative Preference Optimization: Enhancing LLM Alignment through Contrasting Responses across Identical and Diverse Prompts","date":"2024-02-12","arxiv_id":"2402.10958","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yinyueqin/relative-preference-optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"suppressing-pink-elephants-with-direct","title":"Suppressing Pink Elephants with Direct Principle Feedback","date":"2024-02-12","arxiv_id":"2402.07896","n_code_links":0,"syntology":null},{"paper":null,"slug":"v-star-training-verifiers-for-self-taught","title":"V-STaR: Training Verifiers for Self-Taught Reasoners","date":"2024-02-09","arxiv_id":"2402.06457","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-preference-optimization-a-unified","title":"Generalized Preference Optimization: A Unified Approach to Offline Alignment","date":"2024-02-08","arxiv_id":"2402.05749","n_code_links":0,"syntology":null},{"paper":"/paper/noise-contrastive-alignment-of-language","slug":"noise-contrastive-alignment-of-language","title":"Noise Contrastive Alignment of Language Models with Explicit Rewards","date":"2024-02-08","arxiv_id":"2402.05369","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-ml/noise-contrastive-alignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"direct-language-model-alignment-from-online","title":"Direct Language Model Alignment from Online AI Feedback","date":"2024-02-07","arxiv_id":"2402.04792","n_code_links":0,"syntology":null},{"paper":null,"slug":"brain-bayesian-reward-conditioned-amortized","title":"BRAIn: Bayesian Reward-conditioned Amortized Inference for natural language generation from feedback","date":"2024-02-04","arxiv_id":"2402.02479","n_code_links":0,"syntology":null},{"paper":"/paper/lipo-listwise-preference-optimization-through","slug":"lipo-listwise-preference-optimization-through","title":"LiPO: Listwise Preference Optimization through Learning-to-Rank","date":"2024-02-02","arxiv_id":"2402.01878","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/towards-efficient-and-exact-optimization-of","slug":"towards-efficient-and-exact-optimization-of","title":"Towards Efficient Exact Optimization of Language Model Alignment","date":"2024-02-01","arxiv_id":"2402.00856","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haozheji/exact-optimization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-rewarding-language-models","slug":"self-rewarding-language-models","title":"Self-Rewarding Language Models","date":"2024-01-18","arxiv_id":"2401.10020","n_code_links":3,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"aligning-large-language-models-with","title":"Aligning Large Language Models with Counterfactual DPO","date":"2024-01-17","arxiv_id":"2401.09566","n_code_links":0,"syntology":null},{"paper":"/paper/a-mechanistic-understanding-of-alignment","slug":"a-mechanistic-understanding-of-alignment","title":"A Mechanistic Understanding of Alignment Algorithms: A Case Study on DPO and Toxicity","date":"2024-01-03","arxiv_id":"2401.01967","n_code_links":2,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ajyl/dpo_toxic"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"preference-as-reward-maximum-preference","title":"Preference as Reward, Maximum Preference Optimization with Importance Sampling","date":"2023-12-27","arxiv_id":"2312.16430","n_code_links":0,"syntology":null},{"paper":"/paper/some-things-are-more-cringe-than-others","slug":"some-things-are-more-cringe-than-others","title":"Some things are more CRINGE than others: Iterative Preference Optimization with the Pairwise Cringe Loss","date":"2023-12-27","arxiv_id":"2312.16682","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/what-makes-good-data-for-alignment-a","slug":"what-makes-good-data-for-alignment-a","title":"What Makes Good Data for Alignment? A Comprehensive Study of Automatic Data Selection in Instruction Tuning","date":"2023-12-25","arxiv_id":"2312.15685","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hkust-nlp/deita"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gibbs-sampling-from-human-feedback-a-provable","slug":"gibbs-sampling-from-human-feedback-a-provable","title":"Iterative Preference Learning from Human Feedback: Bridging Theory and Practice for RLHF under KL-Constraint","date":"2023-12-18","arxiv_id":"2312.11456","n_code_links":3,"syntology":null},{"paper":"/paper/policy-optimization-in-rlhf-the-impact-of-out","slug":"policy-optimization-in-rlhf-the-impact-of-out","title":"Policy Optimization in RLHF: The Impact of Out-of-preference Data","date":"2023-12-17","arxiv_id":"2312.10584","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["liziniu/policy_optimization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/silkie-preference-distillation-for-large","slug":"silkie-preference-distillation-for-large","title":"Silkie: Preference Distillation for Large Visual Language Models","date":"2023-12-17","arxiv_id":"2312.10665","n_code_links":0,"syntology":null},{"paper":"/paper/ulma-unified-language-model-alignment-with","slug":"ulma-unified-language-model-alignment-with","title":"ULMA: Unified Language Model Alignment with Human Demonstration and Point-wise Preference","date":"2023-12-05","arxiv_id":"2312.02554","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["unified-language-model-alignment/src"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/using-human-feedback-to-fine-tune-diffusion","slug":"using-human-feedback-to-fine-tune-diffusion","title":"Using Human Feedback to Fine-tune Diffusion Models without Any Reward Model","date":"2023-11-22","arxiv_id":"2311.13231","n_code_links":1,"syntology":null},{"paper":"/paper/diffusion-model-alignment-using-direct","slug":"diffusion-model-alignment-using-direct","title":"Diffusion Model Alignment Using Direct Preference Optimization","date":"2023-11-21","arxiv_id":"2311.12908","n_code_links":2,"syntology":{"ran":8,"of":10,"n_ran_checked":4,"n_instrument":4,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/direct-preference-optimization-for-neural","slug":"direct-preference-optimization-for-neural","title":"Direct Preference Optimization for Neural Machine Translation with Minimum Bayes Risk Decoding","date":"2023-11-14","arxiv_id":"2311.08380","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bruceyg/dpo-mbr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/black-box-prompt-optimization-aligning-large","slug":"black-box-prompt-optimization-aligning-large","title":"Black-Box Prompt Optimization: Aligning Large Language Models without Model Training","date":"2023-11-07","arxiv_id":"2311.04155","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-coai/bpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"5a57e8416dd561bffcbe5f6ec59a05d4a9e1df27b5c0702a083ea4110745f1df","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}