{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dpo/papers/ran/1","list_of":"/method/dpo","method":"DPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":112,"counts":{"archive_papers_tagged":409,"with_a_code_link":184,"where_syntology_ran_a_sample":112,"not_listed_spam_title":0,"listed":409,"listed_where_code_ran":112,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":98,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":98,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dpo/papers/ran/1","prev":null,"next":"/method/dpo/papers/ran/2","papers":[{"paper":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":3,"n_instrument":3,"unverified":2,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/tgdpo-harnessing-token-level-reward-guidance","slug":"tgdpo-harnessing-token-level-reward-guidance","title":"TGDPO: Harnessing Token-Level Reward Guidance for Enhancing Direct Preference Optimization","date":"2025-06-17","arxiv_id":"2506.14574","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dvlab-research/tgdpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/levo-high-quality-song-generation-with-multi","slug":"levo-high-quality-song-generation-with-multi","title":"LeVo: High-Quality Song Generation with Multi-Preference Alignment","date":"2025-06-09","arxiv_id":"2506.07520","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tencent-ailab/songgeneration"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"paper":"/paper/superwriter-reflection-driven-long-form","slug":"superwriter-reflection-driven-long-form","title":"SuperWriter: Reflection-Driven Long-Form Generation with Large Language Models","date":"2025-06-04","arxiv_id":"2506.04180","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":2,"n_instrument":4,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mozhu621/superwriter"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/frictional-agent-alignment-framework-slow","slug":"frictional-agent-alignment-framework-slow","title":"Frictional Agent Alignment Framework: Slow Down and Don't Break Things","date":"2025-05-26","arxiv_id":"2505.19428","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["csu-signal/faaf_acl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/token-importance-guided-direct-preference","slug":"token-importance-guided-direct-preference","title":"Token-Importance Guided Direct Preference Optimization","date":"2025-05-26","arxiv_id":"2505.19653","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/understanding-the-performance-gap-in","slug":"understanding-the-performance-gap-in","title":"Understanding the Performance Gap in Preference Learning: A Dichotomy of RLHF and DPO","date":"2025-05-26","arxiv_id":"2505.19770","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["srzer/Gap-in-Preference-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/mpo-multilingual-safety-alignment-via-reward","slug":"mpo-multilingual-safety-alignment-via-reward","title":"MPO: Multilingual Safety Alignment via Reward Gap Optimization","date":"2025-05-22","arxiv_id":"2505.16869","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["circle-hit/mpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mm-ifengine-towards-multimodal-instruction","slug":"mm-ifengine-towards-multimodal-instruction","title":"MM-IFEngine: Towards Multimodal Instruction Following","date":"2025-04-10","arxiv_id":"2504.07957","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["syuan03/mm-ifengine"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/excot-optimizing-reasoning-for-text-to-sql","slug":"excot-optimizing-reasoning-for-text-to-sql","title":"ExCoT: Optimizing Reasoning for Text-to-SQL with Execution Feedback","date":"2025-03-25","arxiv_id":"2503.19988","n_code_links":1,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["snowflakedb/ArcticTraining"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/temple-temporal-preference-learning-of-video","slug":"temple-temporal-preference-learning-of-video","title":"TEMPLE:Temporal Preference Learning of Video LLMs via Difficulty Scheduling and Pre-SFT Alignment","date":"2025-03-21","arxiv_id":"2503.16929","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lscpku/temple"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/light-r1-curriculum-sft-dpo-and-rl-for-long","slug":"light-r1-curriculum-sft-dpo-and-rl-for-long","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","date":"2025-03-13","arxiv_id":"2503.10460","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":8,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["qihoo360/light-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/lightgen-efficient-image-generation-through","slug":"lightgen-efficient-image-generation-through","title":"LightGen: Efficient Image Generation through Knowledge Distillation and Direct Preference Optimization","date":"2025-03-11","arxiv_id":"2503.08619","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":5,"n_instrument":4,"unverified":1,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xianfengwu01/lightgen"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-llm-safety-alignment-with-dual","slug":"improving-llm-safety-alignment-with-dual","title":"Improving LLM Safety Alignment with Dual-Objective Optimization","date":"2025-03-05","arxiv_id":"2503.03710","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wicai24/door-alignment"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-rlhf-be-more-efficient-with-imperfect","slug":"can-rlhf-be-more-efficient-with-imperfect","title":"Can RLHF be More Efficient with Imperfect Reward Models? A Policy Coverage Perspective","date":"2025-02-26","arxiv_id":"2502.19255","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jiaweihhuang/rlhf_rewardtransfer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"paper":"/paper/agentic-reward-modeling-integrating-human","slug":"agentic-reward-modeling-integrating-human","title":"Agentic Reward Modeling: Integrating Human Preferences with Verifiable Correctness Signals for Reliable Reward Systems","date":"2025-02-26","arxiv_id":"2502.19328","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":8,"n_instrument":4,"unverified":2,"pointer_only":0,"phrase":"12 ran (of which 3 constructed an object rather than computing a result; 8 with no instrument failure: 4 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-keg/agentic-reward-modeling"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":3,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/hippo-enhancing-the-table-understanding","slug":"hippo-enhancing-the-table-understanding","title":"HIPPO: Enhancing the Table Understanding Capability of Large Language Models through Hybrid-Modal Preference Optimization","date":"2025-02-24","arxiv_id":"2502.17315","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["neuir/hippo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/earlier-tokens-contribute-more-learning","slug":"earlier-tokens-contribute-more-learning","title":"Earlier Tokens Contribute More: Learning Direct Preference Optimization From Temporal Decay Perspective","date":"2025-02-20","arxiv_id":"2502.14340","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lotusrc/d2po"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/longpo-long-context-self-evolution-of-large","slug":"longpo-long-context-self-evolution-of-large","title":"LongPO: Long Context Self-Evolution of Large Language Models through Short-to-Long Preference Optimization","date":"2025-02-19","arxiv_id":"2502.13922","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DAMO-NLP-SG/LongPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/step-video-t2v-technical-report-the-practice","slug":"step-video-t2v-technical-report-the-practice","title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","date":"2025-02-14","arxiv_id":"2502.10248","n_code_links":3,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stepfun-ai/step-video-t2v"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/principled-data-selection-for-alignment-the","slug":"principled-data-selection-for-alignment-the","title":"Principled Data Selection for Alignment: The Hidden Risks of Difficult Examples","date":"2025-02-11","arxiv_id":"2502.09650","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["glorgao/selectivedpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/guardreasoner-towards-reasoning-based-llm","slug":"guardreasoner-towards-reasoning-based-llm","title":"GuardReasoner: Towards Reasoning-based LLM Safeguards","date":"2025-01-30","arxiv_id":"2501.18492","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yueliu1999/guardreasoner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/wildchat-50m-a-deep-dive-into-the-role-of","slug":"wildchat-50m-a-deep-dive-into-the-role-of","title":"WILDCHAT-50M: A Deep Dive Into the Role of Synthetic Data in Post-Training","date":"2025-01-30","arxiv_id":"2501.18511","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["penfever/wildchat-50m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/chip-cross-modal-hierarchical-direct","slug":"chip-cross-modal-hierarchical-direct","title":"CHiP: Cross-modal Hierarchical Direct Preference Optimization for Multimodal LLMs","date":"2025-01-28","arxiv_id":"2501.16629","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":1,"n_instrument":4,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lvugai/chip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mitigating-hallucinations-in-large-vision-3","slug":"mitigating-hallucinations-in-large-vision-3","title":"Mitigating Hallucinations in Large Vision-Language Models via DPO: On-Policy Data Hold the Key","date":"2025-01-16","arxiv_id":"2501.09695","n_code_links":1,"syntology":{"ran":11,"of":16,"n_ran_checked":9,"n_instrument":2,"unverified":5,"pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["zhyang2226/opa-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/no-preference-left-behind-group","slug":"no-preference-left-behind-group","title":"No Preference Left Behind: Group Distributional Preference Optimization","date":"2024-12-28","arxiv_id":"2412.20299","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["BigBinnie/GDPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/offline-reinforcement-learning-for-llm-multi","slug":"offline-reinforcement-learning-for-llm-multi","title":"Offline Reinforcement Learning for LLM Multi-Step Reasoning","date":"2024-12-20","arxiv_id":"2412.16145","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jwhj/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/preference-oriented-supervised-fine-tuning","slug":"preference-oriented-supervised-fine-tuning","title":"Preference-Oriented Supervised Fine-Tuning: Favoring Target Model Over Aligned Large Language Models","date":"2024-12-17","arxiv_id":"2412.12865","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Savannah120/alignment-handbook-PoFT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sprec-leveraging-self-play-to-debias","slug":"sprec-leveraging-self-play-to-debias","title":"SPRec: Leveraging Self-Play to Debias Preference Alignment for Large Language Model-based Recommendations","date":"2024-12-12","arxiv_id":"2412.09243","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["regionch/sprec"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/patchdpo-patch-level-dpo-for-finetuning-free","slug":"patchdpo-patch-level-dpo-for-finetuning-free","title":"PatchDPO: Patch-level DPO for Finetuning-free Personalized Image Generation","date":"2024-12-04","arxiv_id":"2412.03177","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":8,"n_instrument":2,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hqhqaq/patchdpo"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/critical-tokens-matter-token-level","slug":"critical-tokens-matter-token-level","title":"Critical Tokens Matter: Token-Level Contrastive Estimation Enhances LLM's Reasoning Capability","date":"2024-11-29","arxiv_id":"2411.19943","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chenzhiling9954/critical-tokens-matter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/insight-v-exploring-long-chain-visual","slug":"insight-v-exploring-long-chain-visual","title":"Insight-V: Exploring Long-Chain Visual Reasoning with Multimodal Large Language Models","date":"2024-11-21","arxiv_id":"2411.14432","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":0,"n_instrument":5,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["dongyh20/insight-v"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/ablation-is-not-enough-to-emulate-dpo-how","slug":"ablation-is-not-enough-to-emulate-dpo-how","title":"Beyond Toxic Neurons: A Mechanistic Analysis of DPO for Toxicity Reduction","date":"2024-11-10","arxiv_id":"2411.06424","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yushi-y/dpo-toxic-neurons"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sample-efficient-alignment-for-llms","slug":"sample-efficient-alignment-for-llms","title":"Sample-Efficient Alignment for LLMs","date":"2024-11-03","arxiv_id":"2411.01493","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/oat"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/todo-enhancing-llm-alignment-with-ternary","slug":"todo-enhancing-llm-alignment-with-ternary","title":"TODO: Enhancing LLM Alignment with Ternary Preferences","date":"2024-11-02","arxiv_id":"2411.02442","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["xxares/todo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/vpo-leveraging-the-number-of-votes-in","slug":"vpo-leveraging-the-number-of-votes-in","title":"VPO: Leveraging the Number of Votes in Preference Optimization","date":"2024-10-30","arxiv_id":"2410.22891","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ku-dmlab/vpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/f-po-generalizing-preference-optimization","slug":"f-po-generalizing-preference-optimization","title":"$f$-PO: Generalizing Preference Optimization with $f$-divergence Minimization","date":"2024-10-29","arxiv_id":"2410.21662","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["minkaixu/fpo"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/longreward-improving-long-context-large","slug":"longreward-improving-long-context-large","title":"LongReward: Improving Long-context Large Language Models with AI Feedback","date":"2024-10-28","arxiv_id":"2410.21252","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THUDM/LongReward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/fast-best-of-n-decoding-via-speculative","slug":"fast-best-of-n-decoding-via-speculative","title":"Fast Best-of-N Decoding via Speculative Rejection","date":"2024-10-26","arxiv_id":"2410.20290","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Zanette-Labs/SpeculativeRejection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-dpo-adaptive-reward-margin-is-what-direct","slug":"a-dpo-adaptive-reward-margin-is-what-direct","title":"$α$-DPO: Adaptive Reward Margin is What Direct Preference Optimization Needs","date":"2024-10-14","arxiv_id":"2410.10148","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junkangwu/alpha-dpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-imbalance-driven-rewarding-for","slug":"language-imbalance-driven-rewarding-for","title":"Language Imbalance Driven Rewarding for Multilingual Self-improving","date":"2024-10-11","arxiv_id":"2410.08964","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":6,"n_instrument":2,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["znlp/language-imbalance-driven-rewarding"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/reward-augmented-data-enhances-direct","slug":"reward-augmented-data-enhances-direct","title":"Reward-Augmented Data Enhances Direct Preference Alignment of LLMs","date":"2024-10-10","arxiv_id":"2410.08067","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shenao-zhang/reward-augmented-preference"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tis-dpo-token-level-importance-sampling-for","slug":"tis-dpo-token-level-importance-sampling-for","title":"TIS-DPO: Token-level Importance Sampling for Direct Preference Optimization With Estimated Weights","date":"2024-10-06","arxiv_id":"2410.04350","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["exlaw/TIS-DPO"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/regressing-the-relative-future-efficient","slug":"regressing-the-relative-future-efficient","title":"Regressing the Relative Future: Efficient Policy Optimization for Multi-turn RLHF","date":"2024-10-06","arxiv_id":"2410.04612","n_code_links":1,"syntology":{"ran":14,"of":15,"n_ran_checked":13,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhaolingao/refuel"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/flashmask-efficient-and-rich-mask-extension","slug":"flashmask-efficient-and-rich-mask-extension","title":"FlashMask: Efficient and Rich Mask Extension of FlashAttention","date":"2024-10-02","arxiv_id":"2410.01359","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["PaddlePaddle/Paddle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-crucial-role-of-samplers-in-online-direct","slug":"the-crucial-role-of-samplers-in-online-direct","title":"The Crucial Role of Samplers in Online Direct Preference Optimization","date":"2024-09-29","arxiv_id":"2409.19605","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["srzer/Samplers-in-Online-DPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-extending-direct-preference-optimization","slug":"on-extending-direct-preference-optimization","title":"On Extending Direct Preference Optimization to Accommodate Ties","date":"2024-09-25","arxiv_id":"2409.17431","n_code_links":0,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/instruct-skillmix-a-powerful-pipeline-for-llm","slug":"instruct-skillmix-a-powerful-pipeline-for-llm","title":"Instruct-SkillMix: A Powerful Pipeline for LLM Instruction Tuning","date":"2024-08-27","arxiv_id":"2408.14774","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["princeton-pli/Instruct-SkillMix"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/systematic-evaluation-of-llm-as-a-judge-in","slug":"systematic-evaluation-of-llm-as-a-judge-in","title":"Systematic Evaluation of LLM-as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates","date":"2024-08-23","arxiv_id":"2408.13006","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shenghh2015/llm-judge-eval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/personality-alignment-of-large-language","slug":"personality-alignment-of-large-language","title":"Personality Alignment of Large Language Models","date":"2024-08-21","arxiv_id":"2408.11779","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhu-minjun/palign"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/xgen-mm-blip-3-a-family-of-open-large","slug":"xgen-mm-blip-3-a-family-of-open-large","title":"xGen-MM (BLIP-3): A Family of Open Large Multimodal Models","date":"2024-08-16","arxiv_id":"2408.08872","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/bridging-and-modeling-correlations-in","slug":"bridging-and-modeling-correlations-in","title":"Bridging and Modeling Correlations in Pairwise Data for Direct Preference Optimization","date":"2024-08-14","arxiv_id":"2408.07471","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["YJiangcm/BMC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-training-with-direct-preference","slug":"self-training-with-direct-preference","title":"Self-Training with Direct Preference Optimization Improves Chain-of-Thought Reasoning","date":"2024-07-25","arxiv_id":"2407.18248","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tianduowang/dpo-st"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-reference-policies-in-direct","slug":"understanding-reference-policies-in-direct","title":"Understanding Reference Policies in Direct Preference Optimization","date":"2024-07-18","arxiv_id":"2407.13709","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yale-nlp/refdpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-dynamics-of-llm-finetuning","slug":"learning-dynamics-of-llm-finetuning","title":"Learning Dynamics of LLM Finetuning","date":"2024-07-15","arxiv_id":"2407.10490","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":5,"n_instrument":2,"unverified":3,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["joshua-ren/learning_dynamics_llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/qwen2-audio-technical-report","slug":"qwen2-audio-technical-report","title":"Qwen2-Audio Technical Report","date":"2024-07-15","arxiv_id":"2407.10759","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qwenlm/qwen2-audio"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/b-dpo-direct-preference-optimization-with","slug":"b-dpo-direct-preference-optimization-with","title":"$β$-DPO: Direct Preference Optimization with Dynamic $β$","date":"2024-07-11","arxiv_id":"2407.08639","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junkangwu/beta-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-robust-alignment-of-language-models","slug":"towards-robust-alignment-of-language-models","title":"Towards Robust Alignment of Language Models: Distributionally Robustifying Direct Preference Optimization","date":"2024-07-10","arxiv_id":"2407.07880","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junkangwu/dr_dpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/step-controlled-dpo-leveraging-stepwise-error","slug":"step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","arxiv_id":"2407.00782","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mathllm/Step-Controlled_DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/stllava-med-self-training-large-language-and","slug":"stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","arxiv_id":"2406.19973","n_code_links":1,"syntology":{"ran":9,"of":19,"n_ran_checked":5,"n_instrument":4,"unverified":10,"pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","official":{"repos":["heliossun/stllava-med"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"paper":"/paper/decoding-time-language-model-alignment-with","slug":"decoding-time-language-model-alignment-with","title":"Decoding-Time Language Model Alignment with Multiple Objectives","date":"2024-06-27","arxiv_id":"2406.18853","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["srzer/mod"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/suri-multi-constraint-instruction-following","slug":"suri-multi-constraint-instruction-following","title":"Suri: Multi-constraint Instruction Following for Long-form Text Generation","date":"2024-06-27","arxiv_id":"2406.19371","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":12,"n_instrument":0,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["chtmp223/suri"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":10,"n_instrument":1,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/preference-tuning-for-toxicity-mitigation","slug":"preference-tuning-for-toxicity-mitigation","title":"Preference Tuning For Toxicity Mitigation Generalizes Across Languages","date":"2024-06-23","arxiv_id":"2406.16235","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["batsresearch/cross-lingual-detox"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/direct-multi-turn-preference-optimization-for","slug":"direct-multi-turn-preference-optimization-for","title":"Direct Multi-Turn Preference Optimization for Language Agents","date":"2024-06-21","arxiv_id":"2406.14868","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["swt-user/dmpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-play-with-execution-feedback-improving","slug":"self-play-with-execution-feedback-improving","title":"Self-play with Execution Feedback: Improving Instruction-following Capabilities of Large Language Models","date":"2024-06-19","arxiv_id":"2406.13542","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["QwenLM/AutoIF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dpo-dual-perturbation-optimization-for-test","slug":"dpo-dual-perturbation-optimization-for-test","title":"DPO: Dual-Perturbation Optimization for Test-time Adaptation in 3D Object Detection","date":"2024-06-19","arxiv_id":"2406.13891","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jo-wang/dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/mdpo-conditional-preference-optimization-for","slug":"mdpo-conditional-preference-optimization-for","title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11839","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":2,"n_instrument":3,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":{"repos":["luka-group/mDPO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/is-poisoning-a-real-threat-to-llm-alignment","slug":"is-poisoning-a-real-threat-to-llm-alignment","title":"Is poisoning a real threat to LLM alignment? Maybe more so than you think","date":"2024-06-17","arxiv_id":"2406.12091","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["pankayaraj/RLHFPoisoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/humor-in-ai-massive-scale-crowd-sourced","slug":"humor-in-ai-massive-scale-crowd-sourced","title":"Humor in AI: Massive Scale Crowd-Sourced Preferences and Benchmarks for Cartoon Captioning","date":"2024-06-15","arxiv_id":"2406.10522","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yguooo/cartoon-caption-generation"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-softmax-direct-preference-optimization-for","slug":"on-softmax-direct-preference-optimization-for","title":"On Softmax Direct Preference Optimization for Recommendation","date":"2024-06-13","arxiv_id":"2406.09215","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chenyuxin1999/s-dpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/unpacking-dpo-and-ppo-disentangling-best","slug":"unpacking-dpo-and-ppo-disentangling-best","title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","date":"2024-06-13","arxiv_id":"2406.09279","n_code_links":2,"syntology":{"ran":15,"of":19,"n_ran_checked":8,"n_instrument":7,"unverified":4,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","official":{"repos":["allenai/open-instruct","hamishivi/easylm"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/triple-preference-optimization-achieving","slug":"triple-preference-optimization-achieving","title":"Triple Preference Optimization: Achieving Better Alignment with Less Data in a Single Step Optimization","date":"2024-05-26","arxiv_id":"2405.16681","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sahsaeedi/triple-preference-optimization"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/simpo-simple-preference-optimization-with-a","slug":"simpo-simple-preference-optimization-with-a","title":"SimPO: Simple Preference Optimization with a Reference-Free Reward","date":"2024-05-23","arxiv_id":"2405.14734","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princeton-nlp/simpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/curriculum-direct-preference-optimization-for","slug":"curriculum-direct-preference-optimization-for","title":"Curriculum Direct Preference Optimization for Diffusion and Consistency Models","date":"2024-05-22","arxiv_id":"2405.13637","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":2,"n_instrument":1,"unverified":5,"pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["croitorualin/curriculum-dpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/detox-toxic-subspace-projection-for-model","slug":"detox-toxic-subspace-projection-for-model","title":"Model Editing as a Robust and Denoised variant of DPO: A Case Study on Toxicity","date":"2024-05-22","arxiv_id":"2405.13967","n_code_links":2,"syntology":{"ran":7,"of":12,"n_ran_checked":6,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["uppaal/detox-edit"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/quantifying-and-optimizing-global","slug":"quantifying-and-optimizing-global","title":"Quantifying and Optimizing Global Faithfulness in Persona-driven Role-playing","date":"2024-05-13","arxiv_id":"2405.07726","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["KomeijiForce/Active_Passive_Constraint_Koishiday_2024"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/value-augmented-sampling-for-language-model","slug":"value-augmented-sampling-for-language-model","title":"Value Augmented Sampling for Language Model Alignment and Personalization","date":"2024-05-10","arxiv_id":"2405.06639","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["idanshen/Value-Augmented-Sampling"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/d2po-discriminator-guided-dpo-with-response","slug":"d2po-discriminator-guided-dpo-with-response","title":"D2PO: Discriminator-Guided DPO with Response Evaluation Models","date":"2024-05-02","arxiv_id":"2405.01511","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["PrasannS/d2po"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-play-preference-optimization-for","slug":"self-play-preference-optimization-for","title":"Self-Play Preference Optimization for Language Model Alignment","date":"2024-05-01","arxiv_id":"2405.00675","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["uclaml/sppo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dpo-meets-ppo-reinforced-token-optimization","slug":"dpo-meets-ppo-reinforced-token-optimization","title":"DPO Meets PPO: Reinforced Token Optimization for RLHF","date":"2024-04-29","arxiv_id":"2404.18922","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zkshan2002/rto"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","n_code_links":3,"syntology":{"ran":16,"of":20,"n_ran_checked":12,"n_instrument":4,"unverified":4,"pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/dpo-differential-reinforcement-learning-with","slug":"dpo-differential-reinforcement-learning-with","title":"DPO: A Differential and Pointwise Control Approach to Reinforcement Learning","date":"2024-04-24","arxiv_id":"2404.15617","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/filtered-direct-preference-optimization","slug":"filtered-direct-preference-optimization","title":"Filtered Direct Preference Optimization","date":"2024-04-22","arxiv_id":"2404.13846","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["cyberagentailab/filtered-dpo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/token-level-direct-preference-optimization","slug":"token-level-direct-preference-optimization","title":"Token-level Direct Preference Optimization","date":"2024-04-18","arxiv_id":"2404.11999","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vance0124/token-level-direct-preference-optimization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/regularized-best-of-n-sampling-to-mitigate","slug":"regularized-best-of-n-sampling-to-mitigate","title":"Regularized Best-of-N Sampling with Minimum Bayes Risk Objective for Language Model Alignment","date":"2024-04-01","arxiv_id":"2404.01054","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["CyberAgentAILab/regularized-bon"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/direct-preference-optimization-of-video-large","slug":"direct-preference-optimization-of-video-large","title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","date":"2024-04-01","arxiv_id":"2404.01258","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["riflezhang/llava-hound-dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/comparing-bad-apples-to-good-oranges-aligning","slug":"comparing-bad-apples-to-good-oranges-aligning","title":"Comparing Bad Apples to Good Oranges: Aligning Large Language Models via Joint Preference Optimization","date":"2024-03-31","arxiv_id":"2404.00530","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hritikbansal/dove"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/disentangling-length-from-quality-in-direct","slug":"disentangling-length-from-quality-in-direct","title":"Disentangling Length from Quality in Direct Preference Optimization","date":"2024-03-28","arxiv_id":"2403.19159","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/detoxifying-large-language-models-via","slug":"detoxifying-large-language-models-via","title":"Detoxifying Large Language Models via Knowledge Editing","date":"2024-03-21","arxiv_id":"2403.14472","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/trial-and-error-exploration-based-trajectory","slug":"trial-and-error-exploration-based-trajectory","title":"Trial and Error: Exploration-Based Trajectory Optimization for LLM Agents","date":"2024-03-04","arxiv_id":"2403.02502","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yifan-song793/eto"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/direct-large-language-model-alignment-through","slug":"direct-large-language-model-alignment-through","title":"Direct Large Language Model Alignment Through Self-Rewarding Contrastive Prompt Distillation","date":"2024-02-19","arxiv_id":"2402.11907","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["exlaw/dlma"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/direct-preference-optimization-with-an-offset","slug":"direct-preference-optimization-with-an-offset","title":"Direct Preference Optimization with an Offset","date":"2024-02-16","arxiv_id":"2402.10571","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rycolab/odpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-modal-preference-alignment-remedies","slug":"multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","arxiv_id":"2402.10884","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["findalexli/mllm-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/relative-preference-optimization-enhancing","slug":"relative-preference-optimization-enhancing","title":"Relative Preference Optimization: Enhancing LLM Alignment through Contrasting Responses across Identical and Diverse Prompts","date":"2024-02-12","arxiv_id":"2402.10958","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yinyueqin/relative-preference-optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/noise-contrastive-alignment-of-language","slug":"noise-contrastive-alignment-of-language","title":"Noise Contrastive Alignment of Language Models with Explicit Rewards","date":"2024-02-08","arxiv_id":"2402.05369","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-ml/noise-contrastive-alignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lipo-listwise-preference-optimization-through","slug":"lipo-listwise-preference-optimization-through","title":"LiPO: Listwise Preference Optimization through Learning-to-Rank","date":"2024-02-02","arxiv_id":"2402.01878","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/towards-efficient-and-exact-optimization-of","slug":"towards-efficient-and-exact-optimization-of","title":"Towards Efficient Exact Optimization of Language Model Alignment","date":"2024-02-01","arxiv_id":"2402.00856","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haozheji/exact-optimization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-rewarding-language-models","slug":"self-rewarding-language-models","title":"Self-Rewarding Language Models","date":"2024-01-18","arxiv_id":"2401.10020","n_code_links":3,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}}],"record_sha256":"4e698def2b72a8df77f080e02ca4d276dd1c5d3dac8aefb8cba796b351d18d9d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}