{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-eval-ds-config","entry":"get_eval_ds_config","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":20,"n_papers_ran":14,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":7,"n_samples_fingerprinted":0,"n_places":20,"n_places_pointer_only":9,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":7,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2410.21662","paper":"/paper/f-po-generalizing-preference-optimization","title":"$f$-PO: Generalizing Preference Optimization with $f$-divergence Minimization","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"minkaixu/fpo","path":"src/utils/ds_utils.py","file_url":"https://github.com/minkaixu/fpo/blob/HEAD/src/utils/ds_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"792e87a1badc5642","mcp_get_code":{"code_sha256":"792e87a1badc5642"}},{"arxiv_id":"2410.16843","paper":"/paper/trustworthy-alignment-of-retrieval-augmented","title":"Trustworthy Alignment of Retrieval-Augmented Large Language Models via Reinforcement Learning","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zmzhang2000/trustworthy-alignment","path":"dschat/utils/ds_utils.py","file_url":"https://github.com/zmzhang2000/trustworthy-alignment/blob/HEAD/dschat/utils/ds_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"792e87a1badc5642","mcp_get_code":{"code_sha256":"792e87a1badc5642"}},{"arxiv_id":"2410.07471","paper":"/paper/seal-safety-enhanced-aligned-llm-fine-tuning","title":"SEAL: Safety-enhanced Aligned LLM Fine-tuning via Bilevel Data Selection","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanshen95/seal","path":"seal/utils/deepspeed_utils.py","file_url":"https://github.com/hanshen95/seal/blob/HEAD/seal/utils/deepspeed_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f7c48c9631124114","mcp_get_code":{"code_sha256":"f7c48c9631124114"}},{"arxiv_id":"2408.12109","paper":"/paper/rovrm-a-robust-visual-reward-model-optimized","title":"RoVRM: A Robust Visual Reward Model Optimized via Auxiliary Textual Preference Data","date":"2024-08-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangclnlp/vision-llm-alignment","path":"training/utils/ds_utils.py","file_url":"https://github.com/wangclnlp/vision-llm-alignment/blob/HEAD/training/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ae99b6b77f74c9ae","mcp_get_code":{"code_sha256":"ae99b6b77f74c9ae"}},{"arxiv_id":"2407.16574","paper":"/paper/tlcr-token-level-continuous-reward-for-fine","title":"TLCR: Token-Level Continuous Reward for Fine-grained Reinforcement Learning from Human Feedback","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"esyoon7/rlhf-tlcr","path":"utils/ds_utils.py","file_url":"https://github.com/esyoon7/rlhf-tlcr/blob/HEAD/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e221a76f7d7a3d73","mcp_get_code":{"code_sha256":"e221a76f7d7a3d73"}},{"arxiv_id":"2406.11030","paper":"/paper/foodieqa-a-multimodal-dataset-for-fine","title":"FoodieQA: A Multimodal Dataset for Fine-Grained Understanding of Chinese Food Culture","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lyan62/FoodieQA","path":"Yi/finetune/utils/ds_utils.py","file_url":"https://github.com/lyan62/FoodieQA/blob/HEAD/Yi/finetune/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1182381b66bc654d","mcp_get_code":{"code_sha256":"1182381b66bc654d"}},{"arxiv_id":"2406.10977","paper":"/paper/toward-optimal-llm-alignments-using-two","title":"Toward Optimal LLM Alignments Using Two-Player Games","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ruizheng20/gpo","path":"utils.py","file_url":"https://github.com/ruizheng20/gpo/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"259b0b2755f70e0e","mcp_get_code":{"code_sha256":"259b0b2755f70e0e"}},{"arxiv_id":"2406.08434","paper":"/paper/taste-teaching-large-language-models-to","title":"TasTe: Teaching Large Language Models to Translate through Self-Reflection","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yutongwang1216/reflectionllmmt","path":"train/trainer/utils/ds_utils.py","file_url":"https://github.com/yutongwang1216/reflectionllmmt/blob/HEAD/train/trainer/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1182381b66bc654d","mcp_get_code":{"code_sha256":"1182381b66bc654d"}},{"arxiv_id":"2405.16455","paper":"/paper/on-the-algorithmic-bias-of-aligning-large","title":"On the Algorithmic Bias of Aligning Large Language Models with RLHF: Preference Collapse and Matching Regularization","date":"2024-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JiancongXiao/PM_RLHF","path":"dschat/utils/ds_utils.py","file_url":"https://github.com/JiancongXiao/PM_RLHF/blob/HEAD/dschat/utils/ds_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"792e87a1badc5642","mcp_get_code":{"code_sha256":"792e87a1badc5642"}},{"arxiv_id":"2405.07667","paper":"/paper/backdoor-removal-for-generative-large","title":"Simulate and Eliminate: Revoke Backdoors for Generative Large Language Models","date":"2024-05-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HKUST-KnowComp/SANDE","path":"deepspeed_utils.py","file_url":"https://github.com/HKUST-KnowComp/SANDE/blob/HEAD/deepspeed_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f7c48c9631124114","mcp_get_code":{"code_sha256":"f7c48c9631124114"}},{"arxiv_id":"2404.17287","paper":"/paper/when-to-trust-llms-aligning-confidence-with","title":"When to Trust LLMs: Aligning Confidence with Response Quality","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"taoshuchang/conqord","path":"utils/ds_utils.py","file_url":"https://github.com/taoshuchang/conqord/blob/HEAD/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cfada7d71672fe57","mcp_get_code":{"code_sha256":"cfada7d71672fe57"}},{"arxiv_id":"2403.14112","paper":"/paper/benchmarking-chinese-commonsense-reasoning-of","title":"Benchmarking Chinese Commonsense Reasoning of LLMs: From Chinese-Specifics to Reasoning-Memorization Correlations","date":"2024-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"01-ai/Yi","path":"finetune/utils/ds_utils.py","file_url":"https://github.com/01-ai/Yi/blob/HEAD/finetune/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1182381b66bc654d","mcp_get_code":{"code_sha256":"1182381b66bc654d"}},{"arxiv_id":"2402.18150","paper":"/paper/unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xsc1234/info-rag","path":"utils/ds_utils.py","file_url":"https://github.com/xsc1234/info-rag/blob/HEAD/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1182381b66bc654d","mcp_get_code":{"code_sha256":"1182381b66bc654d"}},{"arxiv_id":"2402.14700","paper":"/paper/unveiling-linguistic-regions-in-large","title":"Unveiling Linguistic Regions in Large Language Models","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zzhang0179/Unveiling-Linguistic-Regions-in-LLMs","path":"training/utils/ds_utils.py","file_url":"https://github.com/zzhang0179/Unveiling-Linguistic-Regions-in-LLMs/blob/HEAD/training/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7f4bdf1898ad0544","mcp_get_code":{"code_sha256":"7f4bdf1898ad0544"}},{"arxiv_id":"2402.05808","paper":"/paper/training-large-language-models-for-reasoning","title":"Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning","date":"2024-02-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"woooodyy/llm-reverse-curriculum-rl","path":"R3_others/dschat/utils/ds_utils.py","file_url":"https://github.com/woooodyy/llm-reverse-curriculum-rl/blob/HEAD/R3_others/dschat/utils/ds_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"792e87a1badc5642","mcp_get_code":{"code_sha256":"792e87a1badc5642"}},{"arxiv_id":"2401.06080","paper":"/paper/secrets-of-rlhf-in-large-language-models-part-1","title":"Secrets of RLHF in Large Language Models Part II: Reward Modeling","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openlmlab/moss-rlhf","path":"utils.py","file_url":"https://github.com/openlmlab/moss-rlhf/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"259b0b2755f70e0e","mcp_get_code":{"code_sha256":"259b0b2755f70e0e"}},{"arxiv_id":"2310.20246","paper":"/paper/breaking-language-barriers-in-multilingual","title":"Breaking Language Barriers in Multilingual Mathematical Reasoning: Insights and Observations","date":"2023-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/MathOctopus","path":"utils/ds_utils.py","file_url":"https://github.com/microsoft/MathOctopus/blob/HEAD/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7f4bdf1898ad0544","mcp_get_code":{"code_sha256":"7f4bdf1898ad0544"}},{"arxiv_id":"2310.10505","paper":"/paper/remax-a-simple-effective-and-efficient-method","title":"ReMax: A Simple, Effective, and Efficient Reinforcement Learning Method for Aligning Large Language Models","date":"2023-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liziniu/ReMax","path":"utils/ds_utils.py","file_url":"https://github.com/liziniu/ReMax/blob/HEAD/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e221a76f7d7a3d73","mcp_get_code":{"code_sha256":"e221a76f7d7a3d73"}},{"arxiv_id":"2308.02223","paper":"/paper/esrl-efficient-sampling-based-reinforcement","title":"ESRL: Efficient Sampling-based Reinforcement Learning for Sequence Generation","date":"2023-08-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangclnlp/DeepSpeed-Chat-Extension","path":"rlhf_llama/deepspeed_chat/training/utils/ds_utils.py","file_url":"https://github.com/wangclnlp/DeepSpeed-Chat-Extension/blob/HEAD/rlhf_llama/deepspeed_chat/training/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8ce22956ed0b989e","mcp_get_code":{"code_sha256":"8ce22956ed0b989e"}},{"arxiv_id":"2024.naacl-long.326","paper":null,"title":"arXiv:2024.naacl-long.326","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LanXiu0523/RLHF_instructGPT","path":"applications/utils/ds_utils.py","file_url":"https://github.com/LanXiu0523/RLHF_instructGPT/blob/HEAD/applications/utils/ds_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1182381b66bc654d","mcp_get_code":{"code_sha256":"1182381b66bc654d"}}]}