{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/only-optimize-lora-parameters","entry":"only_optimize_lora_parameters","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":10,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":2,"n_samples_ran":1,"n_samples_fingerprinted":0,"n_places":10,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2410.21662","paper":"/paper/f-po-generalizing-preference-optimization","title":"$f$-PO: Generalizing Preference Optimization with $f$-divergence Minimization","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"minkaixu/fpo","path":"src/utils/module/lora.py","file_url":"https://github.com/minkaixu/fpo/blob/HEAD/src/utils/module/lora.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7d082cd1b7edc2aa","mcp_get_code":{"code_sha256":"7d082cd1b7edc2aa"}},{"arxiv_id":"2407.16574","paper":"/paper/tlcr-token-level-continuous-reward-for-fine","title":"TLCR: Token-Level Continuous Reward for Fine-grained Reinforcement Learning from Human Feedback","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"esyoon7/rlhf-tlcr","path":"utils/module/lora.py","file_url":"https://github.com/esyoon7/rlhf-tlcr/blob/HEAD/utils/module/lora.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e617041853afd002","mcp_get_code":{"code_sha256":"e617041853afd002"}},{"arxiv_id":"2406.08434","paper":"/paper/taste-teaching-large-language-models-to","title":"TasTe: Teaching Large Language Models to Translate through Self-Reflection","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yutongwang1216/reflectionllmmt","path":"train/trainer/utils/module/lora.py","file_url":"https://github.com/yutongwang1216/reflectionllmmt/blob/HEAD/train/trainer/utils/module/lora.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e617041853afd002","mcp_get_code":{"code_sha256":"e617041853afd002"}},{"arxiv_id":"2405.16455","paper":"/paper/on-the-algorithmic-bias-of-aligning-large","title":"On the Algorithmic Bias of Aligning Large Language Models with RLHF: Preference Collapse and Matching Regularization","date":"2024-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JiancongXiao/PM_RLHF","path":"dschat/utils/module/lora.py","file_url":"https://github.com/JiancongXiao/PM_RLHF/blob/HEAD/dschat/utils/module/lora.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7d082cd1b7edc2aa","mcp_get_code":{"code_sha256":"7d082cd1b7edc2aa"}},{"arxiv_id":"2402.18150","paper":"/paper/unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xsc1234/info-rag","path":"utils/module/lora.py","file_url":"https://github.com/xsc1234/info-rag/blob/HEAD/utils/module/lora.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e617041853afd002","mcp_get_code":{"code_sha256":"e617041853afd002"}},{"arxiv_id":"2402.14700","paper":"/paper/unveiling-linguistic-regions-in-large","title":"Unveiling Linguistic Regions in Large Language Models","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zzhang0179/Unveiling-Linguistic-Regions-in-LLMs","path":"training/utils/module/lora-old.py","file_url":"https://github.com/zzhang0179/Unveiling-Linguistic-Regions-in-LLMs/blob/HEAD/training/utils/module/lora-old.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e617041853afd002","mcp_get_code":{"code_sha256":"e617041853afd002"}},{"arxiv_id":"2402.05808","paper":"/paper/training-large-language-models-for-reasoning","title":"Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning","date":"2024-02-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"woooodyy/llm-reverse-curriculum-rl","path":"R3_others/dschat/utils/module/lora.py","file_url":"https://github.com/woooodyy/llm-reverse-curriculum-rl/blob/HEAD/R3_others/dschat/utils/module/lora.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7d082cd1b7edc2aa","mcp_get_code":{"code_sha256":"7d082cd1b7edc2aa"}},{"arxiv_id":"2310.20246","paper":"/paper/breaking-language-barriers-in-multilingual","title":"Breaking Language Barriers in Multilingual Mathematical Reasoning: Insights and Observations","date":"2023-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/MathOctopus","path":"utils/module/lora.py","file_url":"https://github.com/microsoft/MathOctopus/blob/HEAD/utils/module/lora.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e617041853afd002","mcp_get_code":{"code_sha256":"e617041853afd002"}},{"arxiv_id":"2310.10505","paper":"/paper/remax-a-simple-effective-and-efficient-method","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","date":"2023-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liziniu/ReMax","path":"utils/module/lora.py","file_url":"https://github.com/liziniu/ReMax/blob/HEAD/utils/module/lora.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e617041853afd002","mcp_get_code":{"code_sha256":"e617041853afd002"}},{"arxiv_id":"2024.naacl-long.326","paper":null,"title":"arXiv:2024.naacl-long.326","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LanXiu0523/RLHF_instructGPT","path":"applications/utils/module/lora.py","file_url":"https://github.com/LanXiu0523/RLHF_instructGPT/blob/HEAD/applications/utils/module/lora.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e617041853afd002","mcp_get_code":{"code_sha256":"e617041853afd002"}}]}