{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dpo/papers/3","list_of":"/method/dpo","method":"DPO","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":5,"rows_per_page":100,"rows":[201,300],"of":409,"counts":{"archive_papers_tagged":409,"with_a_code_link":184,"where_syntology_ran_a_sample":112,"not_listed_spam_title":0,"listed":409,"listed_where_code_ran":112,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":98,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":98,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dpo","prev":"/method/dpo/papers/2","next":"/method/dpo/papers/4","papers":[{"paper":"/paper/deformpam-data-efficient-learning-for-long","slug":"deformpam-data-efficient-learning-for-long","title":"DeformPAM: Data-Efficient Learning for Long-horizon Deformable Object Manipulation via Preference-based Action Alignment","date":"2024-10-15","arxiv_id":"2410.11584","n_code_links":1,"syntology":null},{"paper":null,"slug":"have-the-vlms-lost-confidence-a-study-of","title":"Have the VLMs Lost Confidence? A Study of Sycophancy in VLMs","date":"2024-10-15","arxiv_id":"2410.11302","n_code_links":0,"syntology":null},{"paper":"/paper/a-dpo-adaptive-reward-margin-is-what-direct","slug":"a-dpo-adaptive-reward-margin-is-what-direct","title":"$α$-DPO: Adaptive Reward Margin is What Direct Preference Optimization Needs","date":"2024-10-14","arxiv_id":"2410.10148","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junkangwu/alpha-dpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-multi-step-reasoning-abilities-of","slug":"enhancing-multi-step-reasoning-abilities-of","title":"Enhancing Multi-Step Reasoning Abilities of Language Models through Direct Q-Function Optimization","date":"2024-10-11","arxiv_id":"2410.09302","n_code_links":2,"syntology":null},{"paper":"/paper/language-imbalance-driven-rewarding-for","slug":"language-imbalance-driven-rewarding-for","title":"Language Imbalance Driven Rewarding for Multilingual Self-improving","date":"2024-10-11","arxiv_id":"2410.08964","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":6,"n_instrument":2,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["znlp/language-imbalance-driven-rewarding"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"simultaneous-reward-distillation-and","title":"Simultaneous Reward Distillation and Preference Learning: Get You a Language Model Who Can Do Both","date":"2024-10-11","arxiv_id":"2410.08458","n_code_links":0,"syntology":null},{"paper":"/paper/supercorrect-supervising-and-correcting","slug":"supercorrect-supervising-and-correcting","title":"SuperCorrect: Supervising and Correcting Language Models with Error-Driven Insights","date":"2024-10-11","arxiv_id":"2410.09008","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["yangling0818/supercorrect-llm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"evolutionary-contrastive-distillation-for","title":"Evolutionary Contrastive Distillation for Language Model Alignment","date":"2024-10-10","arxiv_id":"2410.07513","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyperdpo-hypernetwork-based-multi-objective","title":"COS-DPO: Conditioned One-Shot Multi-Objective Fine-Tuning Framework","date":"2024-10-10","arxiv_id":"2410.08316","n_code_links":0,"syntology":null},{"paper":null,"slug":"optima-optimizing-effectiveness-and","title":"Optima: Optimizing Effectiveness and Efficiency for LLM-Based Multi-Agent System","date":"2024-10-10","arxiv_id":"2410.08115","n_code_links":0,"syntology":null},{"paper":"/paper/reward-augmented-data-enhances-direct","slug":"reward-augmented-data-enhances-direct","title":"Reward-Augmented Data Enhances Direct Preference Alignment of LLMs","date":"2024-10-10","arxiv_id":"2410.08067","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shenao-zhang/reward-augmented-preference"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tpo-aligning-large-language-models-with-multi","title":"TPO: Aligning Large Language Models with Multi-branch & Multi-step Preference Trees","date":"2024-10-10","arxiv_id":"2410.12854","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-multimodal-llm-for-detailed-and","title":"Enhancing Multimodal LLM for Detailed and Accurate Video Captioning using Multi-Round Preference Optimization","date":"2024-10-09","arxiv_id":"2410.06682","n_code_links":0,"syntology":null},{"paper":null,"slug":"subtle-errors-matter-preference-learning-via","title":"Subtle Errors Matter: Preference Learning via Error-injected Self-editing","date":"2024-10-09","arxiv_id":"2410.06638","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-self-improvement-of-llms-via-mcts","title":"Towards Self-Improvement of LLMs via MCTS: Leveraging Stepwise Knowledge with Curriculum Preference Learning","date":"2024-10-09","arxiv_id":"2410.06508","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerated-preference-optimization-for-large","title":"Accelerated Preference Optimization for Large Language Model Alignment","date":"2024-10-08","arxiv_id":"2410.06293","n_code_links":0,"syntology":null},{"paper":null,"slug":"rlrf4rec-reinforcement-learning-from-recsys","title":"Direct Preference Optimization for LLM-Enhanced Recommendation Systems","date":"2024-10-08","arxiv_id":"2410.05939","n_code_links":0,"syntology":null},{"paper":null,"slug":"as-simple-as-fine-tuning-llm-alignment-via","title":"As Simple as Fine-tuning: LLM Alignment via Bidirectional Negative Feedback Loss","date":"2024-10-07","arxiv_id":"2410.04834","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-rationalization-improves-llm-as-a-fine","title":"Self-rationalization improves LLM as a fine-grained judge","date":"2024-10-07","arxiv_id":"2410.05495","n_code_links":0,"syntology":null},{"paper":"/paper/regressing-the-relative-future-efficient","slug":"regressing-the-relative-future-efficient","title":"Regressing the Relative Future: Efficient Policy Optimization for Multi-turn RLHF","date":"2024-10-06","arxiv_id":"2410.04612","n_code_links":1,"syntology":{"ran":14,"of":15,"n_ran_checked":13,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhaolingao/refuel"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tis-dpo-token-level-importance-sampling-for","slug":"tis-dpo-token-level-importance-sampling-for","title":"TIS-DPO: Token-level Importance Sampling for Direct Preference Optimization With Estimated Weights","date":"2024-10-06","arxiv_id":"2410.04350","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["exlaw/TIS-DPO"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"rainbowpo-a-unified-framework-for-combining","title":"RainbowPO: A Unified Framework for Combining Improvements in Preference Optimization","date":"2024-10-05","arxiv_id":"2410.04203","n_code_links":0,"syntology":null},{"paper":"/paper/an-exploration-of-self-supervised-mutual","slug":"an-exploration-of-self-supervised-mutual","title":"An Exploration of Self-Supervised Mutual Information Alignment for Multi-Task Settings","date":"2024-10-02","arxiv_id":"2410.01704","n_code_links":1,"syntology":null},{"paper":"/paper/flashmask-efficient-and-rich-mask-extension","slug":"flashmask-efficient-and-rich-mask-extension","title":"FlashMask: Efficient and Rich Mask Extension of FlashAttention","date":"2024-10-02","arxiv_id":"2410.01359","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["PaddlePaddle/Paddle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-perfect-blend-redefining-rlhf-with","title":"The Perfect Blend: Redefining RLHF with Mixture of Judges","date":"2024-09-30","arxiv_id":"2409.20370","n_code_links":0,"syntology":null},{"paper":"/paper/the-crucial-role-of-samplers-in-online-direct","slug":"the-crucial-role-of-samplers-in-online-direct","title":"The Crucial Role of Samplers in Online Direct Preference Optimization","date":"2024-09-29","arxiv_id":"2409.19605","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["srzer/Samplers-in-Online-DPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cross-lingual-human-preference-alignment-for","title":"Cross-lingual Human-Preference Alignment for Neural Machine Translation with Direct Quality Optimization","date":"2024-09-26","arxiv_id":"2409.17673","n_code_links":0,"syntology":null},{"paper":null,"slug":"modulated-intervention-preference","title":"Modulated Intervention Preference Optimization (MIPO): Keep the Easy, Refine the Difficult","date":"2024-09-26","arxiv_id":"2409.17545","n_code_links":0,"syntology":null},{"paper":"/paper/on-extending-direct-preference-optimization","slug":"on-extending-direct-preference-optimization","title":"On Extending Direct Preference Optimization to Accommodate Ties","date":"2024-09-25","arxiv_id":"2409.17431","n_code_links":0,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"zeroth-order-policy-gradient-for","title":"Zeroth-Order Policy Gradient for Reinforcement Learning from Human Feedback without Reward Inference","date":"2024-09-25","arxiv_id":"2409.17401","n_code_links":0,"syntology":null},{"paper":null,"slug":"orthogonal-finetuning-for-direct-preference","title":"Orthogonal Finetuning for Direct Preference Optimization","date":"2024-09-23","arxiv_id":"2409.14836","n_code_links":0,"syntology":null},{"paper":null,"slug":"backtracking-improves-generation-safety","title":"Backtracking Improves Generation Safety","date":"2024-09-22","arxiv_id":"2409.14586","n_code_links":0,"syntology":null},{"paper":null,"slug":"rrm-robust-reward-model-training-mitigates","title":"RRM: Robust Reward Model Training Mitigates Reward Hacking","date":"2024-09-20","arxiv_id":"2409.13156","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-large-language-models-for-4","title":"Fine Tuning Large Language Models for Medicine: The Role and Importance of Direct Preference Optimization","date":"2024-09-19","arxiv_id":"2409.12741","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-lists-to-emojis-how-format-bias-affects","title":"From Lists to Emojis: How Format Bias Affects Model Alignment","date":"2024-09-18","arxiv_id":"2409.11704","n_code_links":0,"syntology":null},{"paper":"/paper/asft-aligned-supervised-fine-tuning-through","slug":"asft-aligned-supervised-fine-tuning-through","title":"ASFT: Aligned Supervised Fine-Tuning through Absolute Likelihood","date":"2024-09-14","arxiv_id":"2409.10571","n_code_links":1,"syntology":null},{"paper":null,"slug":"geometric-averaged-preference-optimization","title":"Geometric-Averaged Preference Optimization for Soft Preference Labels","date":"2024-09-10","arxiv_id":"2409.06691","n_code_links":0,"syntology":null},{"paper":null,"slug":"length-desensitization-in-directed-preference","title":"Length Desensitization in Direct Preference Optimization","date":"2024-09-10","arxiv_id":"2409.06411","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-limited-generalization-capability-of","title":"On the Limited Generalization Capability of the Implicit Reward Model Induced by Direct Preference Optimization","date":"2024-09-05","arxiv_id":"2409.03650","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-math-agents-with-multi-turn","title":"Building Math Agents with Multi-Turn Iterative Preference Learning","date":"2024-09-04","arxiv_id":"2409.02392","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-gradient-analysis-framework-for-rewarding","title":"A Gradient Analysis Framework for Rewarding Good and Penalizing Bad Examples in Language Models","date":"2024-08-29","arxiv_id":"2408.16751","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-extremely-data-efficient-and-generative","title":"An Extremely Data-efficient and Generative LLM-based Reinforcement Learning Agent for Recommenders","date":"2024-08-28","arxiv_id":"2408.16032","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-verifiers-reward-modeling-as-next","title":"Generative Verifiers: Reward Modeling as Next-Token Prediction","date":"2024-08-27","arxiv_id":"2408.15240","n_code_links":0,"syntology":null},{"paper":"/paper/instruct-skillmix-a-powerful-pipeline-for-llm","slug":"instruct-skillmix-a-powerful-pipeline-for-llm","title":"Instruct-SkillMix: A Powerful Pipeline for LLM Instruction Tuning","date":"2024-08-27","arxiv_id":"2408.14774","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["princeton-pli/Instruct-SkillMix"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"una-unifying-alignments-of-rlhf-ppo-dpo-and","title":"UNA: Unifying Alignments of RLHF/PPO, DPO and KTO by a Generalized Implicit Reward Function","date":"2024-08-27","arxiv_id":"2408.15339","n_code_links":0,"syntology":null},{"paper":"/paper/systematic-evaluation-of-llm-as-a-judge-in","slug":"systematic-evaluation-of-llm-as-a-judge-in","title":"Systematic Evaluation of LLM-as-a-Judge in LLM Alignment Tasks: Explainable Metrics and Diverse Prompt Templates","date":"2024-08-23","arxiv_id":"2408.13006","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shenghh2015/llm-judge-eval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/personality-alignment-of-large-language","slug":"personality-alignment-of-large-language","title":"Personality Alignment of Large Language Models","date":"2024-08-21","arxiv_id":"2408.11779","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":4,"n_instrument":3,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhu-minjun/palign"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"minor-sft-loss-for-llm-fine-tune-to-increase","title":"Minor SFT loss for LLM fine-tune to increase performance and reduce model deviation","date":"2024-08-20","arxiv_id":"2408.10642","n_code_links":0,"syntology":null},{"paper":null,"slug":"minor-dpo-reject-penalty-to-increase-training","title":"Minor DPO reject penalty to increase training robustness","date":"2024-08-19","arxiv_id":"2408.09834","n_code_links":0,"syntology":null},{"paper":"/paper/xgen-mm-blip-3-a-family-of-open-large","slug":"xgen-mm-blip-3-a-family-of-open-large","title":"xGen-MM (BLIP-3): A Family of Open Large Multimodal Models","date":"2024-08-16","arxiv_id":"2408.08872","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/bridging-and-modeling-correlations-in","slug":"bridging-and-modeling-correlations-in","title":"Bridging and Modeling Correlations in Pairwise Data for Direct Preference Optimization","date":"2024-08-14","arxiv_id":"2408.07471","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["YJiangcm/BMC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/longwriter-unleashing-10000-word-generation","slug":"longwriter-unleashing-10000-word-generation","title":"LongWriter: Unleashing 10,000+ Word Generation from Long Context LLMs","date":"2024-08-13","arxiv_id":"2408.07055","n_code_links":3,"syntology":null},{"paper":"/paper/fuxitranyu-a-multilingual-large-language","slug":"fuxitranyu-a-multilingual-large-language","title":"FuxiTranyu: A Multilingual Large Language Model Trained with Balanced Data","date":"2024-08-12","arxiv_id":"2408.06273","n_code_links":1,"syntology":null},{"paper":"/paper/a-logical-fallacy-informed-framework-for","slug":"a-logical-fallacy-informed-framework-for","title":"A Logical Fallacy-Informed Framework for Argument Generation","date":"2024-08-07","arxiv_id":"2408.03618","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-02923","title":"Intermediate direct preference optimization","date":"2024-08-06","arxiv_id":"2408.02923","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-generalization-of-preference-learning","title":"On the Generalization of Preference Learning with DPO","date":"2024-08-06","arxiv_id":"2408.03459","n_code_links":0,"syntology":null},{"paper":"/paper/self-training-with-direct-preference","slug":"self-training-with-direct-preference","title":"Self-Training with Direct Preference Optimization Improves Chain-of-Thought Reasoning","date":"2024-07-25","arxiv_id":"2407.18248","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tianduowang/dpo-st"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-hitchhiker-s-guide-to-human-alignment","title":"The Hitchhiker's Guide to Human Alignment with *PO","date":"2024-07-21","arxiv_id":"2407.15229","n_code_links":0,"syntology":null},{"paper":null,"slug":"clinical-reading-comprehension-with-encoder","title":"Clinical Reading Comprehension with Encoder-Decoder Models Enhanced by Direct Preference Optimization","date":"2024-07-19","arxiv_id":"2407.14000","n_code_links":0,"syntology":null},{"paper":null,"slug":"decomposed-direct-preference-optimization-for","title":"Decomposed Direct Preference Optimization for Structure-Based Drug Design","date":"2024-07-19","arxiv_id":"2407.13981","n_code_links":0,"syntology":null},{"paper":null,"slug":"correcting-the-mythos-of-kl-regularization","title":"Correcting the Mythos of KL-Regularization: Direct Alignment without Overoptimization via Chi-Squared Preference Optimization","date":"2024-07-18","arxiv_id":"2407.13399","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-reference-policies-in-direct","slug":"understanding-reference-policies-in-direct","title":"Understanding Reference Policies in Direct Preference Optimization","date":"2024-07-18","arxiv_id":"2407.13709","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yale-nlp/refdpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/persllm-a-personified-training-approach-for","slug":"persllm-a-personified-training-approach-for","title":"PersLLM: A Personified Training Approach for Large Language Models","date":"2024-07-17","arxiv_id":"2407.12393","n_code_links":1,"syntology":null},{"paper":"/paper/learning-dynamics-of-llm-finetuning","slug":"learning-dynamics-of-llm-finetuning","title":"Learning Dynamics of LLM Finetuning","date":"2024-07-15","arxiv_id":"2407.10490","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":5,"n_instrument":2,"unverified":3,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["joshua-ren/learning_dynamics_llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/qwen2-audio-technical-report","slug":"qwen2-audio-technical-report","title":"Qwen2-Audio Technical Report","date":"2024-07-15","arxiv_id":"2407.10759","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qwenlm/qwen2-audio"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"new-desiderata-for-direct-preference","title":"New Desiderata for Direct Preference Optimization","date":"2024-07-12","arxiv_id":"2407.09072","n_code_links":0,"syntology":null},{"paper":"/paper/b-dpo-direct-preference-optimization-with","slug":"b-dpo-direct-preference-optimization-with","title":"$β$-DPO: Direct Preference Optimization with Dynamic $β$","date":"2024-07-11","arxiv_id":"2407.08639","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junkangwu/beta-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-robust-alignment-of-language-models","slug":"towards-robust-alignment-of-language-models","title":"Towards Robust Alignment of Language Models: Distributionally Robustifying Direct Preference Optimization","date":"2024-07-10","arxiv_id":"2407.07880","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["junkangwu/dr_dpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-and-accurate-memorable-conversation","title":"Efficient and Accurate Memorable Conversation Model using DPO based on sLLM","date":"2024-07-09","arxiv_id":"2407.06537","n_code_links":0,"syntology":null},{"paper":"/paper/lions-an-empirically-optimized-approach-to","slug":"lions-an-empirically-optimized-approach-to","title":"LIONs: An Empirically Optimized Approach to Align Language Models","date":"2024-07-09","arxiv_id":"2407.06542","n_code_links":1,"syntology":null},{"paper":null,"slug":"exposing-privacy-gaps-membership-inference","title":"Exposing Privacy Gaps: Membership Inference Attack on Preference Data for LLM Alignment","date":"2024-07-08","arxiv_id":"2407.06443","n_code_links":0,"syntology":null},{"paper":"/paper/towards-human-understanding-of-paraphrase","slug":"towards-human-understanding-of-paraphrase","title":"Towards Human Understanding of Paraphrase Types in ChatGPT","date":"2024-07-02","arxiv_id":"2407.02302","n_code_links":3,"syntology":null},{"paper":"/paper/learning-to-explore-and-select-for-coverage","slug":"learning-to-explore-and-select-for-coverage","title":"Learning to Explore and Select for Coverage-Conditioned Retrieval-Augmented Generation","date":"2024-07-01","arxiv_id":"2407.01158","n_code_links":1,"syntology":null},{"paper":null,"slug":"rethinking-llm-based-preference-evaluation","title":"Explaining Length Bias in LLM-Based Preference Evaluations","date":"2024-07-01","arxiv_id":"2407.01085","n_code_links":0,"syntology":null},{"paper":"/paper/step-controlled-dpo-leveraging-stepwise-error","slug":"step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","arxiv_id":"2407.00782","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mathllm/Step-Controlled_DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/stllava-med-self-training-large-language-and","slug":"stllava-med-self-training-large-language-and","title":"STLLaVA-Med: Self-Training Large Language and Vision Assistant for Medical Question-Answering","date":"2024-06-28","arxiv_id":"2406.19973","n_code_links":1,"syntology":{"ran":9,"of":19,"n_ran_checked":5,"n_instrument":4,"unverified":10,"pointer_only":0,"phrase":"9 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","official":{"repos":["heliossun/stllava-med"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":10,"ran_from_kinds":["official"]}}},{"paper":"/paper/decoding-time-language-model-alignment-with","slug":"decoding-time-language-model-alignment-with","title":"Decoding-Time Language Model Alignment with Multiple Objectives","date":"2024-06-27","arxiv_id":"2406.18853","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["srzer/mod"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/suri-multi-constraint-instruction-following","slug":"suri-multi-constraint-instruction-following","title":"Suri: Multi-constraint Instruction Following for Long-form Text Generation","date":"2024-06-27","arxiv_id":"2406.19371","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":12,"n_instrument":0,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["chtmp223/suri"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":10,"n_instrument":1,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-to-ask-informative-questions","slug":"learning-to-ask-informative-questions","title":"Learning to Ask Informative Questions: Enhancing LLMs with Preference Optimization and Expected Information Gain","date":"2024-06-25","arxiv_id":"2406.17453","n_code_links":1,"syntology":{"ran":0,"of":5,"n_ran_checked":0,"n_instrument":0,"unverified":5,"pointer_only":5,"phrase":"0 ran · 5 unverified","official":{"repos":["dmazzaccara/learningtoask"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"paper":null,"slug":"not-all-preference-pairs-are-created-equal-a","title":"Not All Preference Pairs Are Created Equal: A Recipe for Annotation-Efficient Iterative Preference Learning","date":"2024-06-25","arxiv_id":"2406.17312","n_code_links":0,"syntology":null},{"paper":null,"slug":"paft-a-parallel-training-paradigm-for","title":"PAFT: A Parallel Training Paradigm for Effective LLM Fine-Tuning","date":"2024-06-25","arxiv_id":"2406.17923","n_code_links":0,"syntology":null},{"paper":"/paper/preference-tuning-for-toxicity-mitigation","slug":"preference-tuning-for-toxicity-mitigation","title":"Preference Tuning For Toxicity Mitigation Generalizes Across Languages","date":"2024-06-23","arxiv_id":"2406.16235","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["batsresearch/cross-lingual-detox"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/direct-multi-turn-preference-optimization-for","slug":"direct-multi-turn-preference-optimization-for","title":"Direct Multi-Turn Preference Optimization for Language Agents","date":"2024-06-21","arxiv_id":"2406.14868","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["swt-user/dmpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sail-self-improving-efficient-online","title":"SAIL: Self-Improving Efficient Online Alignment of Large Language Models","date":"2024-06-21","arxiv_id":"2406.15567","n_code_links":0,"syntology":null},{"paper":"/paper/dpo-dual-perturbation-optimization-for-test","slug":"dpo-dual-perturbation-optimization-for-test","title":"DPO: Dual-Perturbation Optimization for Test-time Adaptation in 3D Object Detection","date":"2024-06-19","arxiv_id":"2406.13891","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jo-wang/dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-play-with-execution-feedback-improving","slug":"self-play-with-execution-feedback-improving","title":"Self-play with Execution Feedback: Improving Instruction-following Capabilities of Large Language Models","date":"2024-06-19","arxiv_id":"2406.13542","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["QwenLM/AutoIF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/is-poisoning-a-real-threat-to-llm-alignment","slug":"is-poisoning-a-real-threat-to-llm-alignment","title":"Is poisoning a real threat to LLM alignment? Maybe more so than you think","date":"2024-06-17","arxiv_id":"2406.12091","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["pankayaraj/RLHFPoisoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"iterative-length-regularized-direct","title":"Iterative Length-Regularized Direct Preference Optimization: A Case Study on Improving 7B Language Models to GPT-4 Level","date":"2024-06-17","arxiv_id":"2406.11817","n_code_links":0,"syntology":null},{"paper":"/paper/mdpo-conditional-preference-optimization-for","slug":"mdpo-conditional-preference-optimization-for","title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11839","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":2,"n_instrument":3,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":{"repos":["luka-group/mDPO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/eliminating-biased-length-reliance-of-direct","slug":"eliminating-biased-length-reliance-of-direct","title":"Eliminating Biased Length Reliance of Direct Preference Optimization via Down-Sampled KL Divergence","date":"2024-06-16","arxiv_id":"2406.10957","n_code_links":1,"syntology":null},{"paper":"/paper/humor-in-ai-massive-scale-crowd-sourced","slug":"humor-in-ai-massive-scale-crowd-sourced","title":"Humor in AI: Massive Scale Crowd-Sourced Preferences and Benchmarks for Cartoon Captioning","date":"2024-06-15","arxiv_id":"2406.10522","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yguooo/cartoon-caption-generation"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/bootstrapping-language-models-with-dpo","slug":"bootstrapping-language-models-with-dpo","title":"Bootstrapping Language Models with DPO Implicit Rewards","date":"2024-06-14","arxiv_id":"2406.09760","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["sail-sg/dice"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"knowledge-editing-in-language-models-via","title":"Knowledge Editing in Language Models via Adapted Direct Preference Optimization","date":"2024-06-14","arxiv_id":"2406.09920","n_code_links":0,"syntology":null},{"paper":"/paper/on-softmax-direct-preference-optimization-for","slug":"on-softmax-direct-preference-optimization-for","title":"On Softmax Direct Preference Optimization for Recommendation","date":"2024-06-13","arxiv_id":"2406.09215","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chenyuxin1999/s-dpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/unpacking-dpo-and-ppo-disentangling-best","slug":"unpacking-dpo-and-ppo-disentangling-best","title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","date":"2024-06-13","arxiv_id":"2406.09279","n_code_links":2,"syntology":{"ran":15,"of":19,"n_ran_checked":8,"n_instrument":7,"unverified":4,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","official":{"repos":["allenai/open-instruct","hamishivi/easylm"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"3d-properties-identifying-challenges-in-dpo","title":"3D-Properties: Identifying Challenges in DPO and Charting a Path Forward","date":"2024-06-11","arxiv_id":"2406.07327","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-demonstration-and-preference-learning","title":"Learning Reward and Policy Jointly from Demonstration and Preference Improves Alignment","date":"2024-06-11","arxiv_id":"2406.06874","n_code_links":0,"syntology":null},{"paper":null,"slug":"direct-preference-optimization-for","title":"Direct Preference Optimization for Suppressing Hallucinated Prior Exams in Radiology Report Generation","date":"2024-06-10","arxiv_id":"2406.06496","n_code_links":0,"syntology":null},{"paper":null,"slug":"margin-aware-preference-optimization-for","title":"Margin-aware Preference Optimization for Aligning Diffusion Models without Reference","date":"2024-06-10","arxiv_id":"2406.06424","n_code_links":0,"syntology":null}],"record_sha256":"281660ee29248ba2b0afb01d42beec9974582876f4f908ba217556f20c8f4e50","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}