{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/eval-policy","entry":"eval_policy","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":15,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":12,"n_samples_ran":5,"n_samples_fingerprinted":0,"n_places":15,"n_places_pointer_only":5,"by_status":{"ran_honours":4,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":0,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2603.12087","paper":"/paper/arxiv-2603-12087","title":"Cross-Domain Policy Optimization via Bellman Consistency and Hybrid Critics","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"dmksjfl/PAR","path":"train_offline.py","file_url":"https://github.com/dmksjfl/PAR/blob/HEAD/train_offline.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2a67e91feeab6fb8","mcp_get_code":{"code_sha256":"2a67e91feeab6fb8"}},{"arxiv_id":"2506.08460","paper":"/paper/2506-08460","title":"MOBODY: Model Based Off-Dynamics Offline Reinforcement Learning","date":"2025-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guoyihonggyh/mobody-model-based-off-dynamics-offline-reinforcement-learning","path":"train_mobody.py","file_url":"https://github.com/guoyihonggyh/mobody-model-based-off-dynamics-offline-reinforcement-learning/blob/HEAD/train_mobody.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1dbd30f3f1fdb2d2","mcp_get_code":{"code_sha256":"1dbd30f3f1fdb2d2"}},{"arxiv_id":"2410.20750","paper":"/paper/odrl-a-benchmark-for-off-dynamics","title":"ODRL: A Benchmark for Off-Dynamics Reinforcement Learning","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"offdynamicsrl/off-dynamics-rl","path":"train_tune.py","file_url":"https://github.com/offdynamicsrl/off-dynamics-rl/blob/HEAD/train_tune.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2a67e91feeab6fb8","mcp_get_code":{"code_sha256":"2a67e91feeab6fb8"}},{"arxiv_id":"2405.17476","paper":"/paper/how-to-leverage-diverse-demonstrations-in","title":"How to Leverage Diverse Demonstrations in Offline Imitation Learning","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ryanxhr/DWBC","path":"main_setting_demodice.py","file_url":"https://github.com/ryanxhr/DWBC/blob/HEAD/main_setting_demodice.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8205eb63bd6bd52d","mcp_get_code":{"code_sha256":"8205eb63bd6bd52d"}},{"arxiv_id":"2405.15369","paper":"/paper/cross-domain-policy-adaptation-by-capturing","title":"Cross-Domain Policy Adaptation by Capturing Representation Mismatch","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmksjfl/par","path":"train_offline.py","file_url":"https://github.com/dmksjfl/par/blob/HEAD/train_offline.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2a67e91feeab6fb8","mcp_get_code":{"code_sha256":"2a67e91feeab6fb8"}},{"arxiv_id":"2404.07148","paper":"/paper/how-consistent-are-clinicians-evaluating-the","title":"How Consistent are Clinicians? Evaluating the Predictability of Sepsis Disease Progression with Dynamics Models","date":"2024-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cmudig/ai-clinician-mimiciv","path":"ai_clinician/modeling/models/discrete_bcq.py","file_url":"https://github.com/cmudig/ai-clinician-mimiciv/blob/HEAD/ai_clinician/modeling/models/discrete_bcq.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c7422e3be6dd4cd4","mcp_get_code":{"code_sha256":"c7422e3be6dd4cd4"}},{"arxiv_id":"2312.17248","paper":"/paper/rethinking-model-based-policy-based-and-value","title":"Rethinking Model-based, Policy-based, and Value-based Reinforcement Learning via the Lens of Representation Complexity","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guhfeng/rl-representation-complexity","path":"train_TD3.py","file_url":"https://github.com/guhfeng/rl-representation-complexity/blob/HEAD/train_TD3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5a60c90aa154c9b6","mcp_get_code":{"code_sha256":"5a60c90aa154c9b6"}},{"arxiv_id":"2210.13846","paper":"/paper/adaptive-behavior-cloning-regularization-for-1","title":"Adaptive Behavior Cloning Regularization for Stable Offline-to-Online Reinforcement Learning","date":"2022-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaoyi11/adaptive_bc","path":"utils.py","file_url":"https://github.com/zhaoyi11/adaptive_bc/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9823315d1dac2e28","mcp_get_code":{"code_sha256":"9823315d1dac2e28"}},{"arxiv_id":"2203.08949","paper":"/paper/latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pcchenxi/lapo-offlienrl","path":"main_d4rl.py","file_url":"https://github.com/pcchenxi/lapo-offlienrl/blob/HEAD/main_d4rl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5738a20931777f8d","mcp_get_code":{"code_sha256":"5738a20931777f8d"}},{"arxiv_id":"2111.12673","paper":"/paper/adaptively-calibrated-critic-estimates-for","title":"Adaptively Calibrated Critic Estimates for Deep Reinforcement Learning","date":"2021-11-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nicolinho/acc","path":"tqc/functions.py","file_url":"https://github.com/nicolinho/acc/blob/HEAD/tqc/functions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"984c06514cd95568","mcp_get_code":{"code_sha256":"984c06514cd95568"}},{"arxiv_id":"2010.15920","paper":"/paper/recovery-rl-safe-reinforcement-learning-with","title":"Recovery RL: Safe Reinforcement Learning with Learned Recovery Zones","date":"2020-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zlr20/saferl_kit","path":"saferl_algos/recovery.py","file_url":"https://github.com/zlr20/saferl_kit/blob/HEAD/saferl_algos/recovery.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9380f9c95fc0dfaf","mcp_get_code":{"code_sha256":"9380f9c95fc0dfaf"}},{"arxiv_id":"2010.09776","paper":"/paper/smarts-scalable-multi-agent-reinforcement","title":"SMARTS: Scalable Multi-Agent Reinforcement Learning Training School for Autonomous Driving","date":"2020-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mcederle99/MAD4QN-PS","path":"util_rgb.py","file_url":"https://github.com/mcederle99/MAD4QN-PS/blob/HEAD/util_rgb.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"99c73d29a74932a2","mcp_get_code":{"code_sha256":"99c73d29a74932a2"}},{"arxiv_id":"2007.02040","paper":"/paper/discount-factor-as-a-regularizer-in","title":"Discount Factor as a Regularizer in Reinforcement Learning","date":"2020-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ron-amit/Discount_as_Regularizer","path":"TD3_Code/mainTD3.py","file_url":"https://github.com/ron-amit/Discount_as_Regularizer/blob/HEAD/TD3_Code/mainTD3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5a60c90aa154c9b6","mcp_get_code":{"code_sha256":"5a60c90aa154c9b6"}},{"arxiv_id":"2006.06555","paper":"/paper/distributed-reinforcement-learning-in-multi","title":"Multi-Agent Reinforcement Learning in Stochastic Networked Systems","date":"2020-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yihenglin97/networked-marl","path":"WirelessAccess/neural_q_access_training.py","file_url":"https://github.com/yihenglin97/networked-marl/blob/HEAD/WirelessAccess/neural_q_access_training.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"68b5e7e7c1792139","mcp_get_code":{"code_sha256":"68b5e7e7c1792139"}},{"arxiv_id":"1801.08757","paper":"/paper/safe-exploration-in-continuous-action-spaces","title":"Safe Exploration in Continuous Action Spaces","date":"2018-01-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zlr20/saferl_kit","path":"saferl_algos/safetylayer.py","file_url":"https://github.com/zlr20/saferl_kit/blob/HEAD/saferl_algos/safetylayer.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c1793cbf39af800c","mcp_get_code":{"code_sha256":"c1793cbf39af800c"}}]}