{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-reward","entry":"get_reward","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":20,"n_papers_ran":10,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":19,"n_samples_ran":9,"n_samples_fingerprinted":1,"n_places":22,"n_places_pointer_only":8,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":1,"ran":4,"unverified":10},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2506.20629","paper":"/paper/plop-precise-lora-placement-for-efficient","title":"PLoP: Precise LoRA Placement for Efficient Finetuning of Large Models","date":"2025-06-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"soufiane001/plop","path":"sft/model_utils.py","file_url":"https://github.com/soufiane001/plop/blob/HEAD/sft/model_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c24b405d6da63aab","mcp_get_code":{"code_sha256":"c24b405d6da63aab"}},{"arxiv_id":"2506.08691","paper":"/paper/vrest-enhancing-reasoning-in-large-vision","title":"VReST: Enhancing Reasoning in Large Vision-Language Models through Tree Search and Self-Reward Mechanism","date":"2025-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GaryJiajia/VReST","path":"prompt_methods/ours/our6.py","file_url":"https://github.com/GaryJiajia/VReST/blob/HEAD/prompt_methods/ours/our6.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c78a0abbbc546ffb","mcp_get_code":{"code_sha256":"c78a0abbbc546ffb"}},{"arxiv_id":"2505.19770","paper":"/paper/understanding-the-performance-gap-in","title":"Understanding the Performance Gap in Preference Learning: A Dichotomy of RLHF and DPO","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"srzer/Gap-in-Preference-Learning","path":"Exp-0/annotate_data/get_rewards.py","file_url":"https://github.com/srzer/Gap-in-Preference-Learning/blob/HEAD/Exp-0/annotate_data/get_rewards.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0d39de6cae03583a","mcp_get_code":{"code_sha256":"0d39de6cae03583a"}},{"arxiv_id":"2503.21295","paper":"/paper/r-prm-reasoning-driven-process-reward","title":"R-PRM: Reasoning-Driven Process Reward Modeling","date":"2025-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"njunlp/r-prm","path":"src/reward-guide-search/code/GenPRM-Greedy-Seach-Zero-Shot.py","file_url":"https://github.com/njunlp/r-prm/blob/HEAD/src/reward-guide-search/code/GenPRM-Greedy-Seach-Zero-Shot.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9b43f9ef6cfe06cf","mcp_get_code":{"code_sha256":"9b43f9ef6cfe06cf"}},{"arxiv_id":"2503.13538","paper":"/paper/from-demonstrations-to-rewards-alignment","title":"From Demonstrations to Rewards: Alignment Without Explicit Human Preferences","date":"2025-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Hong-Lab-UMN-ECE/IRLAlignment","path":"summarize_from_feedback_details/IRL_reward.py","file_url":"https://github.com/Hong-Lab-UMN-ECE/IRLAlignment/blob/HEAD/summarize_from_feedback_details/IRL_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c64a8d32a7ed56ca","mcp_get_code":{"code_sha256":"c64a8d32a7ed56ca"}},{"arxiv_id":"2502.16182","paper":"/paper/ipo-your-language-model-is-secretly-a","title":"IPO: Your Language Model is Secretly a Preference Classifier","date":"2025-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shivank21/Implicit_Preference_Optimization","path":"Preference_Comparison/RM_Bench/reward_model.py","file_url":"https://github.com/shivank21/Implicit_Preference_Optimization/blob/HEAD/Preference_Comparison/RM_Bench/reward_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"be3a2b51c3f2cfc0","mcp_get_code":{"code_sha256":"be3a2b51c3f2cfc0"}},{"arxiv_id":"2502.16182","paper":"/paper/ipo-your-language-model-is-secretly-a","title":"IPO: Your Language Model is Secretly a Preference Classifier","date":"2025-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shivank21/Implicit_Preference_Optimization","path":"Preference_Comparison/Reward_Bench/reward_model.py","file_url":"https://github.com/shivank21/Implicit_Preference_Optimization/blob/HEAD/Preference_Comparison/Reward_Bench/reward_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"499246b2aadf2aa3","mcp_get_code":{"code_sha256":"499246b2aadf2aa3"}},{"arxiv_id":"2406.17098","paper":"/paper/learning-temporal-distances-contrastive","title":"Learning Temporal Distances: Contrastive Successor Features Can Provide a Metric Structure for Decision-Making","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mnm-anonymous/qmd","path":"envs/fetch_envs.py","file_url":"https://github.com/mnm-anonymous/qmd/blob/HEAD/envs/fetch_envs.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"35bd6ebd08652e20","mcp_get_code":{"code_sha256":"35bd6ebd08652e20"}},{"arxiv_id":"2406.07971","paper":"/paper/it-takes-two-on-the-seamlessness-between","title":"It Takes Two: On the Seamlessness between Reward and Policy Model in RLHF","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"taiminglu/seamless","path":"code/SEAM/reward/get_reward.py","file_url":"https://github.com/taiminglu/seamless/blob/HEAD/code/SEAM/reward/get_reward.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"646f36a115dc20f4","mcp_get_code":{"code_sha256":"646f36a115dc20f4"}},{"arxiv_id":"2404.16767","paper":"/paper/rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaolingao/rebel","path":"src/tldr/rebel.py","file_url":"https://github.com/zhaolingao/rebel/blob/HEAD/src/tldr/rebel.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"724222affceacea6","mcp_get_code":{"code_sha256":"724222affceacea6"}},{"arxiv_id":"2404.16767","paper":"/paper/rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaolingao/rebel","path":"src/tldr/rm.py","file_url":"https://github.com/zhaolingao/rebel/blob/HEAD/src/tldr/rm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c02add7716c931ba","mcp_get_code":{"code_sha256":"c02add7716c931ba"}},{"arxiv_id":"2404.08495","paper":"/paper/dataset-reset-policy-optimization-for-rlhf","title":"Dataset Reset Policy Optimization for RLHF","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cornell-rl/drpo","path":"src/tldr/drpo.py","file_url":"https://github.com/cornell-rl/drpo/blob/HEAD/src/tldr/drpo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"724222affceacea6","mcp_get_code":{"code_sha256":"724222affceacea6"}},{"arxiv_id":"2403.17031","paper":"/paper/the-n-implementation-details-of-rlhf-with-ppo","title":"The N+ Implementation Details of RLHF with PPO: A Case Study on TL;DR Summarization","date":"2024-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vwxyzjn/summarize_from_feedback_details","path":"summarize_from_feedback_details/reward.py","file_url":"https://github.com/vwxyzjn/summarize_from_feedback_details/blob/HEAD/summarize_from_feedback_details/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c64a8d32a7ed56ca","mcp_get_code":{"code_sha256":"c64a8d32a7ed56ca"}},{"arxiv_id":"2402.00782","paper":"/paper/dense-reward-for-free-in-reinforcement","title":"Dense Reward for Free in Reinforcement Learning from Human Feedback","date":"2024-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xanderjc/attention-based-credit","path":"abcrl/reward_modelling/starling.py","file_url":"https://github.com/xanderjc/attention-based-credit/blob/HEAD/abcrl/reward_modelling/starling.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6d632dad98d9d43e","mcp_get_code":{"code_sha256":"6d632dad98d9d43e"}},{"arxiv_id":"2310.20141","paper":"/paper/contrastive-difference-predictive-coding","title":"Contrastive Difference Predictive Coding","date":"2023-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chongyi-zheng/td_infonce","path":"envs/fetch_envs.py","file_url":"https://github.com/chongyi-zheng/td_infonce/blob/HEAD/envs/fetch_envs.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"35bd6ebd08652e20","mcp_get_code":{"code_sha256":"35bd6ebd08652e20"}},{"arxiv_id":"2310.07287","paper":"/paper/interactive-interior-design-recommendation","title":"Interactive Interior Design Recommendation via Coarse-to-fine Multimodal Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2023-10","title_source":"archive","repo":"zhhe11/iidrs","path":"recommendationReady/interact_rl.py","file_url":"https://github.com/zhhe11/iidrs/blob/HEAD/recommendationReady/interact_rl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cdd14236b5d5dafa","mcp_get_code":{"code_sha256":"cdd14236b5d5dafa"}},{"arxiv_id":"2210.03930","paper":"/paper/hierarchical-graph-transformer-with-adaptive","title":"Hierarchical Graph Transformer with Adaptive Node Sampling","date":"2022-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zaixizhang/ans-gt","path":"main_adaptive.py","file_url":"https://github.com/zaixizhang/ans-gt/blob/HEAD/main_adaptive.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"11da82521241501c","mcp_get_code":{"code_sha256":"11da82521241501c"}},{"arxiv_id":"2106.02190","paper":"/paper/spatial-graph-attention-and-curiosity-driven","title":"Spatial Graph Attention and Curiosity-driven Policy for Antiviral Drug Discovery","date":"2021-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yulun-rayn/dgapn","path":"dgapn/reward/get_reward.py","file_url":"https://github.com/yulun-rayn/dgapn/blob/HEAD/dgapn/reward/get_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d7caa82df22437b8","mcp_get_code":{"code_sha256":"d7caa82df22437b8"}},{"arxiv_id":"2106.01325","paper":"/paper/addressing-the-long-term-impact-of-ml","title":"Addressing the Long-term Impact of ML Decisions via Policy Regret","date":"2021-06-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"david-lindner/single-peaked-bandits","path":"src/single_peaked_bandits/fico_experiment/preprocessing.py","file_url":"https://github.com/david-lindner/single-peaked-bandits/blob/HEAD/src/single_peaked_bandits/fico_experiment/preprocessing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"aac98b644db36e6f","mcp_get_code":{"code_sha256":"aac98b644db36e6f"}},{"arxiv_id":"2009.09988","paper":"/paper/robust-outlier-arm-identification-1","title":"Robust Outlier Arm Identification","date":"2020-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yinglunz/ROAI_ICML2020","path":"ROAI_class.py","file_url":"https://github.com/yinglunz/ROAI_ICML2020/blob/HEAD/ROAI_class.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"232905e0b4a83116","mcp_get_code":{"code_sha256":"232905e0b4a83116"}},{"arxiv_id":"2006.10460","paper":"/paper/confident-off-policy-evaluation-and-selection","title":"Confident Off-Policy Evaluation and Selection through Self-Normalized Importance Weighting","date":"2020-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepmind/offpolicy_selection_eslb","path":"data.py","file_url":"https://github.com/deepmind/offpolicy_selection_eslb/blob/HEAD/data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c5467ffc43de8b9d","mcp_get_code":{"code_sha256":"c5467ffc43de8b9d"}},{"arxiv_id":"1810.00821","paper":"/paper/variational-discriminator-bottleneck","title":"Variational Discriminator Bottleneck: Improving Imitation Learning, Inverse RL, and GANs by Constraining Information Flow","date":"2018-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"reinforcement-learning-kr/lets-do-irl","path":"mountaincar/maxent/maxent.py","file_url":"https://github.com/reinforcement-learning-kr/lets-do-irl/blob/HEAD/mountaincar/maxent/maxent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ac908e366f291aaa","mcp_get_code":{"code_sha256":"ac908e366f291aaa"}}]}