{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sft/papers/4","list_of":"/method/sft","method":"SFT","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":415,"counts":{"archive_papers_tagged":415,"with_a_code_link":204,"where_syntology_ran_a_sample":103,"not_listed_spam_title":0,"listed":415,"listed_where_code_ran":103,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sft","prev":"/method/sft/papers/3","next":"/method/sft/papers/5","papers":[{"paper":"/paper/a-fine-tuning-dataset-and-benchmark-for-large","slug":"a-fine-tuning-dataset-and-benchmark-for-large","title":"A Fine-tuning Dataset and Benchmark for Large Language Models for Protein Understanding","date":"2024-06-08","arxiv_id":"2406.05540","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tsynbio/proteinlmdataset"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/extroversion-or-introversion-controlling-the","slug":"extroversion-or-introversion-controlling-the","title":"Extroversion or Introversion? Controlling The Personality of Your Large Language Models","date":"2024-06-07","arxiv_id":"2406.04583","n_code_links":1,"syntology":null},{"paper":null,"slug":"proofread-fixes-all-errors-with-one-tap","title":"Proofread: Fixes All Errors with One Tap","date":"2024-06-06","arxiv_id":"2406.04523","n_code_links":0,"syntology":null},{"paper":"/paper/parrot-multilingual-visual-instruction-tuning","slug":"parrot-multilingual-visual-instruction-tuning","title":"Parrot: Multilingual Visual Instruction Tuning","date":"2024-06-04","arxiv_id":"2406.02539","n_code_links":2,"syntology":{"ran":3,"of":7,"n_ran_checked":0,"n_instrument":3,"unverified":4,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["aidc-ai/parrot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/strengthened-symbol-binding-makes-large","slug":"strengthened-symbol-binding-makes-large","title":"Strengthened Symbol Binding Makes Large Language Models Reliable Multiple-Choice Selectors","date":"2024-06-03","arxiv_id":"2406.01026","n_code_links":1,"syntology":null},{"paper":null,"slug":"longskywork-a-training-recipe-for-efficiently","title":"LongSkywork: A Training Recipe for Efficiently Extending Context Length in Large Language Models","date":"2024-06-02","arxiv_id":"2406.00605","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-practice-friendly-two-stage-llm-enhanced","title":"A Practice-Friendly LLM-Enhanced Paradigm with Preference Parsing for Sequential Recommendation","date":"2024-06-01","arxiv_id":"2406.00333","n_code_links":0,"syntology":null},{"paper":null,"slug":"instructioncp-a-fast-approach-to-transfer","title":"InstructionCP: A fast approach to transfer Large Language Models into target language","date":"2024-05-30","arxiv_id":"2405.20175","n_code_links":0,"syntology":null},{"paper":"/paper/pediatricsgpt-large-language-models-as","slug":"pediatricsgpt-large-language-models-as","title":"PediatricsGPT: Large Language Models as Chinese Medical Assistants for Pediatric Applications","date":"2024-05-29","arxiv_id":"2405.19266","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ydk122024/pediatricsgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/getting-more-juice-out-of-the-sft-data-reward","slug":"getting-more-juice-out-of-the-sft-data-reward","title":"Getting More Juice Out of the SFT Data: Reward Learning from Human Demonstration Improves SFT for LLM Alignment","date":"2024-05-28","arxiv_id":"2405.17888","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":2,"n_instrument":1,"unverified":5,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["jasonjiaxiangli/reward_learning_sft"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/online-merging-optimizers-for-boosting","slug":"online-merging-optimizers-for-boosting","title":"Online Merging Optimizers for Boosting Rewards and Mitigating Tax in Alignment","date":"2024-05-28","arxiv_id":"2405.17931","n_code_links":1,"syntology":null},{"paper":null,"slug":"training-llms-to-better-self-debug-and","title":"LeDex: Training LLMs to Better Self-Debug and Explain Code","date":"2024-05-28","arxiv_id":"2405.18649","n_code_links":0,"syntology":null},{"paper":"/paper/bayesian-rg-flow-in-neural-network-field","slug":"bayesian-rg-flow-in-neural-network-field","title":"Bayesian RG Flow in Neural Network Field Theories","date":"2024-05-27","arxiv_id":"2405.17538","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-modal-safety-alignment-is-textual","title":"Cross-Modal Safety Alignment: Is textual unlearning all you need?","date":"2024-05-27","arxiv_id":"2406.02575","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-llm-journey-from-cognition-to","title":"Exploring the LLM Journey from Cognition to Expression with Linear Representations","date":"2024-05-27","arxiv_id":"2405.16964","n_code_links":0,"syntology":null},{"paper":"/paper/automatically-generating-numerous-context","slug":"automatically-generating-numerous-context","title":"Automatically Generating Numerous Context-Driven SFT Data for LLMs across Diverse Granularity","date":"2024-05-26","arxiv_id":"2405.16579","n_code_links":1,"syntology":null},{"paper":null,"slug":"provably-mitigating-overoptimization-in-rlhf","title":"Provably Mitigating Overoptimization in RLHF: Your SFT Loss is Implicitly an Adversarial Regularizer","date":"2024-05-26","arxiv_id":"2405.16436","n_code_links":0,"syntology":null},{"paper":"/paper/triple-preference-optimization-achieving","slug":"triple-preference-optimization-achieving","title":"Triple Preference Optimization: Achieving Better Alignment with Less Data in a Single Step Optimization","date":"2024-05-26","arxiv_id":"2405.16681","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["sahsaeedi/triple-preference-optimization"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multilingual-prosody-transfer-comparing","title":"Multilingual Prosody Transfer: Comparing Supervised & Transfer Learning","date":"2024-05-23","arxiv_id":"2406.00022","n_code_links":0,"syntology":null},{"paper":"/paper/360zhinao-technical-report","slug":"360zhinao-technical-report","title":"360Zhinao Technical Report","date":"2024-05-22","arxiv_id":"2405.13386","n_code_links":1,"syntology":null},{"paper":"/paper/disperse-then-merge-pushing-the-limits-of","slug":"disperse-then-merge-pushing-the-limits-of","title":"Disperse-Then-Merge: Pushing the Limits of Instruction Tuning via Alignment Tax Reduction","date":"2024-05-22","arxiv_id":"2405.13432","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["TingchenFu/ACL24-ExpertFusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/intuitive-fine-tuning-towards-unifying-sft","slug":"intuitive-fine-tuning-towards-unifying-sft","title":"Intuitive Fine-Tuning: Towards Simplifying Alignment into a Single Process","date":"2024-05-20","arxiv_id":"2405.11870","n_code_links":1,"syntology":null},{"paper":null,"slug":"rethinking-overlooked-aspects-in-vision","title":"Rethinking Overlooked Aspects in Vision-Language Models","date":"2024-05-20","arxiv_id":"2405.11850","n_code_links":0,"syntology":null},{"paper":null,"slug":"advanced-natural-based-interaction-for-the","title":"Advanced Natural-based interaction for the ITAlian language: LLaMAntino-3-ANITA","date":"2024-05-11","arxiv_id":"2405.07101","n_code_links":0,"syntology":null},{"paper":"/paper/a-lightweight-transformer-for-remote-sensing","slug":"a-lightweight-transformer-for-remote-sensing","title":"A Lightweight Sparse Focus Transformer for Remote Sensing Image Change Captioning","date":"2024-05-10","arxiv_id":"2405.06598","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-language-model-s-reasoning","title":"Optimizing Language Model's Reasoning Abilities with Weak Supervision","date":"2024-05-07","arxiv_id":"2405.04086","n_code_links":0,"syntology":null},{"paper":"/paper/huixiangdou-cr-coreference-resolution-in","slug":"huixiangdou-cr-coreference-resolution-in","title":"Labeling supervised fine-tuning data with the scaling law","date":"2024-05-05","arxiv_id":"2405.02817","n_code_links":2,"syntology":null},{"paper":null,"slug":"flame-factuality-aware-alignment-for-large","title":"FLAME: Factuality-Aware Alignment for Large Language Models","date":"2024-05-02","arxiv_id":"2405.01525","n_code_links":0,"syntology":null},{"paper":null,"slug":"prefix-text-as-a-yarn-eliciting-non-english","title":"Prefix Text as a Yarn: Eliciting Non-English Alignment in Foundation Language Model","date":"2024-04-25","arxiv_id":"2404.16766","n_code_links":0,"syntology":null},{"paper":"/paper/weak-to-strong-extrapolation-expedites","slug":"weak-to-strong-extrapolation-expedites","title":"Weak-to-Strong Extrapolation Expedites Alignment","date":"2024-04-25","arxiv_id":"2404.16792","n_code_links":1,"syntology":null},{"paper":"/paper/insights-into-alignment-evaluating-dpo-and","slug":"insights-into-alignment-evaluating-dpo-and","title":"Insights into Alignment: Evaluating DPO and its Variants Across Multiple Tasks","date":"2024-04-23","arxiv_id":"2404.14723","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-preference-driven-paradigm-for-enhanced","title":"A Preference-driven Paradigm for Enhanced Translation with Large Language Models","date":"2024-04-17","arxiv_id":"2404.11288","n_code_links":0,"syntology":null},{"paper":"/paper/balancing-speciality-and-versatility-a-coarse","slug":"balancing-speciality-and-versatility-a-coarse","title":"Balancing Speciality and Versatility: a Coarse to Fine Framework for Supervised Fine-tuning Large Language Model","date":"2024-04-16","arxiv_id":"2404.10306","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rattlesnakey/cofitune"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learn-your-reference-model-for-real-good","title":"Learn Your Reference Model for Real Good Alignment","date":"2024-04-15","arxiv_id":"2404.09656","n_code_links":0,"syntology":null},{"paper":null,"slug":"chinese-tiny-llm-pretraining-a-chinese","title":"Chinese Tiny LLM: Pretraining a Chinese-Centric Large Language Model","date":"2024-04-05","arxiv_id":"2404.04167","n_code_links":0,"syntology":null},{"paper":"/paper/token-efficient-leverage-learning-in-large","slug":"token-efficient-leverage-learning-in-large","title":"Token-Efficient Leverage Learning in Large Language Models","date":"2024-04-01","arxiv_id":"2404.00914","n_code_links":1,"syntology":null},{"paper":"/paper/extensive-self-contrast-enables-feedback-free","slug":"extensive-self-contrast-enables-feedback-free","title":"Extensive Self-Contrast Enables Feedback-Free Language Model Alignment","date":"2024-03-31","arxiv_id":"2404.00604","n_code_links":2,"syntology":null},{"paper":null,"slug":"dialectical-alignment-resolving-the-tension","title":"Dialectical Alignment: Resolving the Tension of 3H and Security Threats of LLMs","date":"2024-03-30","arxiv_id":"2404.00486","n_code_links":0,"syntology":null},{"paper":null,"slug":"injecting-new-knowledge-into-large-language","title":"Injecting New Knowledge into Large Language Models via Supervised Fine-Tuning","date":"2024-03-30","arxiv_id":"2404.00213","n_code_links":0,"syntology":null},{"paper":"/paper/detoxifying-large-language-models-via","slug":"detoxifying-large-language-models-via","title":"Detoxifying Large Language Models via Knowledge Editing","date":"2024-03-21","arxiv_id":"2403.14472","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-novel-paradigm-boosting-translation","title":"A Novel Paradigm Boosting Translation Capabilities of Large Language Models","date":"2024-03-18","arxiv_id":"2403.11430","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-probabilistic-approach-for-alignment-with","title":"A Probabilistic Approach for Model Alignment with Human Comparisons","date":"2024-03-16","arxiv_id":"2403.10771","n_code_links":0,"syntology":null},{"paper":"/paper/reference-free-monolithic-preference","slug":"reference-free-monolithic-preference","title":"ORPO: Monolithic Preference Optimization without Reference Model","date":"2024-03-12","arxiv_id":"2403.07691","n_code_links":4,"syntology":{"ran":11,"of":14,"n_ran_checked":10,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"11 ran (of which 1 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["xfactlab/orpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/smalltolarge-s2l-scalable-data-selection-for","slug":"smalltolarge-s2l-scalable-data-selection-for","title":"SmallToLarge (S2L): Scalable Data Selection for Fine-tuning Large Language Models by Summarizing Training Trajectories of Small Models","date":"2024-03-12","arxiv_id":"2403.07384","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bigml-cs-ucla/s2l"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/unfamiliar-finetuning-examples-control-how","slug":"unfamiliar-finetuning-examples-control-how","title":"Unfamiliar Finetuning Examples Control How Language Models Hallucinate","date":"2024-03-08","arxiv_id":"2403.05612","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["katiekang1998/llm_hallucinations"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/common-7b-language-models-already-possess","slug":"common-7b-language-models-already-possess","title":"Common 7B Language Models Already Possess Strong Math Capabilities","date":"2024-03-07","arxiv_id":"2403.04706","n_code_links":2,"syntology":{"ran":21,"of":28,"n_ran_checked":20,"n_instrument":1,"unverified":7,"pointer_only":12,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["xwin-lm/xwin-lm"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"teaching-large-language-models-to-reason-with","title":"Teaching Large Language Models to Reason with Reinforcement Learning","date":"2024-03-07","arxiv_id":"2403.04642","n_code_links":0,"syntology":null},{"paper":null,"slug":"balancing-enhancement-harmlessness-and","title":"Balancing Enhancement, Harmlessness, and General Capabilities: Enhancing Conversational LLMs with Direct RLHF","date":"2024-03-04","arxiv_id":"2403.02513","n_code_links":0,"syntology":null},{"paper":"/paper/daco-towards-application-driven-and","slug":"daco-towards-application-driven-and","title":"DACO: Towards Application-Driven and Comprehensive Data Analysis via Code Generation","date":"2024-03-04","arxiv_id":"2403.02528","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shirley-wu/daco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"provably-robust-dpo-aligning-language-models","title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","date":"2024-03-01","arxiv_id":"2403.00409","n_code_links":0,"syntology":null},{"paper":"/paper/reflect-rl-two-player-online-rl-fine-tuning","slug":"reflect-rl-two-player-online-rl-fine-tuning","title":"Reflect-RL: Two-Player Online RL Fine-Tuning for LMs","date":"2024-02-20","arxiv_id":"2402.12621","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":1,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zhourunlong/reflect-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-critical-evaluation-of-ai-feedback-for","slug":"a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["architsharma97/dpo-rlaif"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"rs-dpo-a-hybrid-rejection-sampling-and-direct","title":"RS-DPO: A Hybrid Rejection Sampling and Direct Preference Optimization Method for Alignment of Large Language Models","date":"2024-02-15","arxiv_id":"2402.10038","n_code_links":0,"syntology":null},{"paper":"/paper/icdpo-effectively-borrowing-alignment","slug":"icdpo-effectively-borrowing-alignment","title":"ICDPO: Effectively Borrowing Alignment Capability of Others via In-context Direct Preference Optimization","date":"2024-02-14","arxiv_id":"2402.09320","n_code_links":1,"syntology":null},{"paper":"/paper/entgpt-linking-generative-large-language","slug":"entgpt-linking-generative-large-language","title":"EntGPT: Linking Generative Large Language Models with Knowledge Bases","date":"2024-02-09","arxiv_id":"2402.06738","n_code_links":1,"syntology":null},{"paper":null,"slug":"rethinking-data-selection-for-supervised-fine","title":"Rethinking Data Selection for Supervised Fine-Tuning","date":"2024-02-08","arxiv_id":"2402.06094","n_code_links":0,"syntology":null},{"paper":"/paper/pedagogical-alignment-of-large-language","slug":"pedagogical-alignment-of-large-language","title":"Pedagogical Alignment of Large Language Models","date":"2024-02-07","arxiv_id":"2402.05000","n_code_links":1,"syntology":null},{"paper":"/paper/ultralink-an-open-source-knowledge-enhanced","slug":"ultralink-an-open-source-knowledge-enhanced","title":"UltraLink: An Open-Source Knowledge-Enhanced Multilingual Supervised Fine-tuning Dataset","date":"2024-02-07","arxiv_id":"2402.04588","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["openbmb/ultralink"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/anls-a-universal-document-processing-metric","slug":"anls-a-universal-document-processing-metric","title":"ANLS* -- A Universal Document Processing Metric for Generative Large Language Models","date":"2024-02-06","arxiv_id":"2402.03848","n_code_links":2,"syntology":null},{"paper":"/paper/tuning-large-multimodal-models-for-videos","slug":"tuning-large-multimodal-models-for-videos","title":"Tuning Large Multimodal Models for Videos using Reinforcement Learning from AI Feedback","date":"2024-02-06","arxiv_id":"2402.03746","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-political-preferences-of-llms","title":"The Political Preferences of LLMs","date":"2024-02-02","arxiv_id":"2402.01789","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-sparse-fine-tuning-to-large-language","slug":"scaling-sparse-fine-tuning-to-large-language","title":"Scaling Sparse Fine-Tuning to Large Language Models","date":"2024-01-29","arxiv_id":"2401.16405","n_code_links":2,"syntology":{"ran":13,"of":15,"n_ran_checked":9,"n_instrument":4,"unverified":2,"pointer_only":5,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alanansell/peft","ducdauge/sft-llm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"yoda-teacher-student-progressive-learning-for","title":"YODA: Teacher-Student Progressive Learning for Language Models","date":"2024-01-28","arxiv_id":"2401.15670","n_code_links":0,"syntology":null},{"paper":null,"slug":"equipping-language-models-with-tool-use","title":"Equipping Language Models with Tool Use Capability for Tabular Data Analysis in Finance","date":"2024-01-27","arxiv_id":"2401.15328","n_code_links":0,"syntology":null},{"paper":"/paper/supervised-fine-tuning-in-turn-improves","slug":"supervised-fine-tuning-in-turn-improves","title":"Supervised Fine-tuning in turn Improves Visual Foundation Models","date":"2024-01-18","arxiv_id":"2401.10222","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["tencentarc/visft"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/reft-reasoning-with-reinforced-fine-tuning","slug":"reft-reasoning-with-reinforced-fine-tuning","title":"ReFT: Reasoning with Reinforced Fine-Tuning","date":"2024-01-17","arxiv_id":"2401.08967","n_code_links":1,"syntology":{"ran":5,"of":14,"n_ran_checked":5,"n_instrument":0,"unverified":9,"pointer_only":13,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["lqtrung1998/mwp_reft"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrastive-preference-optimization-pushing","slug":"contrastive-preference-optimization-pushing","title":"Contrastive Preference Optimization: Pushing the Boundaries of LLM Performance in Machine Translation","date":"2024-01-16","arxiv_id":"2401.08417","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":4,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fe1ixxu/alma"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/extending-llms-context-window-with-100","slug":"extending-llms-context-window-with-100","title":"Extending LLMs' Context Window with 100 Samples","date":"2024-01-13","arxiv_id":"2401.07004","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gair-nlp/entropy-abf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-experimental-design-framework-for-label","title":"An Experimental Design Framework for Label-Efficient Supervised Finetuning of Large Language Models","date":"2024-01-12","arxiv_id":"2401.06692","n_code_links":0,"syntology":null},{"paper":"/paper/self-play-fine-tuning-converts-weak-language","slug":"self-play-fine-tuning-converts-weak-language","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","date":"2024-01-02","arxiv_id":"2401.01335","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uclaml/SPIN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/geogalactica-a-scientific-large-language","slug":"geogalactica-a-scientific-large-language","title":"GeoGalactica: A Scientific Large Language Model in Geoscience","date":"2023-12-31","arxiv_id":"2401.00434","n_code_links":1,"syntology":null},{"paper":null,"slug":"laffi-leveraging-hybrid-natural-language","title":"LaFFi: Leveraging Hybrid Natural Language Feedback for Fine-tuning Language Models","date":"2023-12-31","arxiv_id":"2401.00907","n_code_links":0,"syntology":null},{"paper":"/paper/advancing-ttp-analysis-harnessing-the-power","slug":"advancing-ttp-analysis-harnessing-the-power","title":"Advancing TTP Analysis: Harnessing the Power of Large Language Models with Retrieval Augmented Generation","date":"2023-12-30","arxiv_id":"2401.00280","n_code_links":1,"syntology":null},{"paper":"/paper/what-makes-good-data-for-alignment-a","slug":"what-makes-good-data-for-alignment-a","title":"What Makes Good Data for Alignment? A Comprehensive Study of Automatic Data Selection in Instruction Tuning","date":"2023-12-25","arxiv_id":"2312.15685","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hkust-nlp/deita"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"carve3d-improving-multi-view-reconstruction","title":"Carve3D: Improving Multi-view Reconstruction Consistency for Diffusion Models with RL Finetuning","date":"2023-12-21","arxiv_id":"2312.13980","n_code_links":0,"syntology":null},{"paper":"/paper/on-task-performance-and-model-calibration","slug":"on-task-performance-and-model-calibration","title":"On Task Performance and Model Calibration with Supervised and Self-Ensembled In-Context Learning","date":"2023-12-21","arxiv_id":"2312.13772","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["cambridgeltl/ensembled-sicl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/huref-human-readable-fingerprint-for-large","slug":"huref-human-readable-fingerprint-for-large","title":"HuRef: HUman-REadable Fingerprint for Large Language Models","date":"2023-12-08","arxiv_id":"2312.04828","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lumia-group/huref"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-unlocking-spell-on-base-llms-rethinking","slug":"the-unlocking-spell-on-base-llms-rethinking","title":"The Unlocking Spell on Base LLMs: Rethinking Alignment via In-Context Learning","date":"2023-12-04","arxiv_id":"2312.01552","n_code_links":1,"syntology":null},{"paper":"/paper/rankinggpt-empowering-large-language-models","slug":"rankinggpt-empowering-large-language-models","title":"A Two-Stage Adaptation of Large Language Models for Text Ranking","date":"2023-11-28","arxiv_id":"2311.16720","n_code_links":1,"syntology":null},{"paper":"/paper/sharegpt4v-improving-large-multi-modal-models","slug":"sharegpt4v-improving-large-multi-modal-models","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","date":"2023-11-21","arxiv_id":"2311.12793","n_code_links":1,"syntology":null},{"paper":null,"slug":"examining-modularity-in-multilingual-lms-via","title":"Examining Modularity in Multilingual LMs via Language-Specialized Subnetworks","date":"2023-11-14","arxiv_id":"2311.08273","n_code_links":0,"syntology":null},{"paper":"/paper/chimed-gpt-a-chinese-medical-large-language","slug":"chimed-gpt-a-chinese-medical-large-language","title":"ChiMed-GPT: A Chinese Medical Large Language Model with Full Training Regime and Better Alignment to Human Preferences","date":"2023-11-10","arxiv_id":"2311.06025","n_code_links":1,"syntology":null},{"paper":"/paper/beyond-imitation-leveraging-fine-grained","slug":"beyond-imitation-leveraging-fine-grained","title":"Beyond Imitation: Leveraging Fine-grained Quality Signals for Alignment","date":"2023-11-07","arxiv_id":"2311.04072","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rucaibox/figa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-models-are-super-mario-absorbing","slug":"language-models-are-super-mario-absorbing","title":"Language Models are Super Mario: Absorbing Abilities from Homologous Models as a Free Lunch","date":"2023-11-06","arxiv_id":"2311.03099","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yule-buaa/mergelm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tailoring-self-rationalizers-with-multi","slug":"tailoring-self-rationalizers-with-multi","title":"Tailoring Self-Rationalizers with Multi-Reward Distillation","date":"2023-11-06","arxiv_id":"2311.02805","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ink-usc/rationalemultirewarddistillation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/vanishing-gradients-in-reinforcement","slug":"vanishing-gradients-in-reinforcement","title":"Vanishing Gradients in Reinforcement Finetuning of Language Models","date":"2023-10-31","arxiv_id":"2310.20703","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apple/ml-rlgrad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/superhf-supervised-iterative-learning-from","slug":"superhf-supervised-iterative-learning-from","title":"SuperHF: Supervised Iterative Learning from Human Feedback","date":"2023-10-25","arxiv_id":"2310.16763","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["openfeedback/superhf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"lobass-gauging-learnability-in-supervised","title":"DavIR: Data Selection via Implicit Reward for Large Language Models","date":"2023-10-16","arxiv_id":"2310.13008","n_code_links":0,"syntology":null},{"paper":"/paper/qilin-med-multi-stage-knowledge-injection","slug":"qilin-med-multi-stage-knowledge-injection","title":"Qilin-Med: Multi-stage Knowledge Injection Advanced Medical Large Language Model","date":"2023-10-13","arxiv_id":"2310.09089","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":1,"n_instrument":4,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["williamliujl/Qilin-Med"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/understanding-the-effects-of-rlhf-on-llm","slug":"understanding-the-effects-of-rlhf-on-llm","title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity","date":"2023-10-10","arxiv_id":"2310.06452","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["facebookresearch/rlfh-gen-div"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/how-abilities-in-large-language-models-are","slug":"how-abilities-in-large-language-models-are","title":"How Abilities in Large Language Models are Affected by Supervised Fine-tuning Data Composition","date":"2023-10-09","arxiv_id":"2310.05492","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ofa-sys/gsm8k-screl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/reinforcement-learning-in-the-era-of-llms","slug":"reinforcement-learning-in-the-era-of-llms","title":"Reinforcement Learning in the Era of LLMs: What is Essential? What is needed? An RL Perspective on RLHF, Prompting, and Beyond","date":"2023-10-09","arxiv_id":"2310.06147","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/chat-vector-a-simple-approach-to-equip-llms","slug":"chat-vector-a-simple-approach-to-equip-llms","title":"Chat Vector: A Simple Approach to Equip LLMs with Instruction Following and Model Alignment in New Languages","date":"2023-10-07","arxiv_id":"2310.04799","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"citing-large-language-models-create","title":"CITING: Large Language Models Create Curriculum for Instruction Tuning","date":"2023-10-04","arxiv_id":"2310.02527","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-post-training-large-language","slug":"contrastive-post-training-large-language","title":"Automatic Pair Construction for Contrastive Post-training","date":"2023-10-03","arxiv_id":"2310.02263","n_code_links":1,"syntology":null},{"paper":"/paper/gpt-fathom-benchmarking-large-language-models","slug":"gpt-fathom-benchmarking-large-language-models","title":"GPT-Fathom: Benchmarking Large Language Models to Decipher the Evolutionary Path towards GPT-4 and Beyond","date":"2023-09-28","arxiv_id":"2309.16583","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gpt-fathom/gpt-fathom"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/rrhf-rank-responses-to-align-language-models-1","slug":"rrhf-rank-responses-to-align-language-models-1","title":"RRHF: Rank Responses to Align Language Models with Human Feedback","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"statistical-rejection-sampling-improves","title":"Statistical Rejection Sampling Improves Preference Optimization","date":"2023-09-13","arxiv_id":"2309.06657","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-autoregressive-multi-modal-models","slug":"scaling-autoregressive-multi-modal-models","title":"Scaling Autoregressive Multi-Modal Models: Pretraining and Instruction Tuning","date":"2023-09-05","arxiv_id":"2309.02591","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-rlhf-reducing-the-memory-usage-of","title":"Efficient RLHF: Reducing the Memory Usage of PPO","date":"2023-09-01","arxiv_id":"2309.00754","n_code_links":0,"syntology":null}],"record_sha256":"b4e72d151b56d874dffec26f4d241ff76d63c9ca4fd3cbdabaefedfa5855efd0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}