{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/reward","entry":"reward","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":13,"n_papers_ran":5,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":17,"n_samples_ran":5,"n_samples_fingerprinted":2,"n_places":17,"n_places_pointer_only":10,"by_status":{"ran_honours":3,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":1,"ran":1,"unverified":12},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.18864","paper":"/paper/arxiv-2605-18864","title":"SAGE: Shaping Anchors for Guided Exploration in RLVR of LLMs","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"tally0818/SAGE","path":"src/train/rewards.py","file_url":"https://github.com/tally0818/SAGE/blob/HEAD/src/train/rewards.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"85a2bb1a8e77c2eb","mcp_get_code":{"code_sha256":"85a2bb1a8e77c2eb"}},{"arxiv_id":"2603.27884","paper":"/paper/arxiv-2603-27884","title":"Near-Optimal Primal-Dual Algorithm for Learning Linear Mixture CMDPs with Adversarial Rewards","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"kihyun-yu/pd-powers","path":"cmdp_primal_dual_power.py","file_url":"https://github.com/kihyun-yu/pd-powers/blob/HEAD/cmdp_primal_dual_power.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b608dceacbfac8c9","mcp_get_code":{"code_sha256":"b608dceacbfac8c9"}},{"arxiv_id":"2602.13807","paper":"/paper/arxiv-2602-13807","title":"AnomaMind: Agentic Time Series Anomaly Detection with Tool-Augmented Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Xiaoyu-Tao/AnomaMind-TS","path":"src/AnomaMind/core/reward.py","file_url":"https://github.com/Xiaoyu-Tao/AnomaMind-TS/blob/HEAD/src/AnomaMind/core/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"5ed7ae732e853625","mcp_get_code":{"code_sha256":"5ed7ae732e853625"}},{"arxiv_id":"2411.14251","paper":"/paper/natural-language-reinforcement-learning-1","title":"Natural Language Reinforcement Learning","date":"2024-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"waterhorse1/natural-language-rl","path":"nlrl/evaluate.py","file_url":"https://github.com/waterhorse1/natural-language-rl/blob/HEAD/nlrl/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a2863dd7dd994a0a","mcp_get_code":{"code_sha256":"a2863dd7dd994a0a"}},{"arxiv_id":"2312.14000","paper":"/paper/risk-sensitive-stochastic-optimal-control-as","title":"Risk-Sensitive Stochastic Optimal Control as Rao-Blackwellized Markovian Score Climbing","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanyas/psoc","path":"psoc/environments/feedback/acrobot_env.py","file_url":"https://github.com/hanyas/psoc/blob/HEAD/psoc/environments/feedback/acrobot_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fb02748e906d2ebf","mcp_get_code":{"code_sha256":"fb02748e906d2ebf"}},{"arxiv_id":"2312.14000","paper":"/paper/risk-sensitive-stochastic-optimal-control-as","title":"Risk-Sensitive Stochastic Optimal Control as Rao-Blackwellized Markovian Score Climbing","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanyas/psoc","path":"psoc/environments/feedback/cartpole_env.py","file_url":"https://github.com/hanyas/psoc/blob/HEAD/psoc/environments/feedback/cartpole_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"793237d1ad68a705","mcp_get_code":{"code_sha256":"793237d1ad68a705"}},{"arxiv_id":"2312.14000","paper":"/paper/risk-sensitive-stochastic-optimal-control-as","title":"Risk-Sensitive Stochastic Optimal Control as Rao-Blackwellized Markovian Score Climbing","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanyas/psoc","path":"psoc/environments/feedback/const_linear_env.py","file_url":"https://github.com/hanyas/psoc/blob/HEAD/psoc/environments/feedback/const_linear_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7e7dcbf5a8571d7c","mcp_get_code":{"code_sha256":"7e7dcbf5a8571d7c"}},{"arxiv_id":"2312.14000","paper":"/paper/risk-sensitive-stochastic-optimal-control-as","title":"Risk-Sensitive Stochastic Optimal Control as Rao-Blackwellized Markovian Score Climbing","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanyas/psoc","path":"psoc/environments/feedback/double_pendulum_env.py","file_url":"https://github.com/hanyas/psoc/blob/HEAD/psoc/environments/feedback/double_pendulum_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8b0f817f54e161c5","mcp_get_code":{"code_sha256":"8b0f817f54e161c5"}},{"arxiv_id":"2312.13912","paper":"/paper/solving-long-run-average-reward-robust-mdps","title":"Solving Long-run Average Reward Robust MDPs via Stochastic Games","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mehrdad76/rmdp-lra","path":"contamination.py","file_url":"https://github.com/mehrdad76/rmdp-lra/blob/HEAD/contamination.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8627a3a38478b85f","mcp_get_code":{"code_sha256":"8627a3a38478b85f"}},{"arxiv_id":"2306.00629","paper":"/paper/identifiability-and-generalizability-in","title":"Identifiability and Generalizability in Constrained Inverse Reinforcement Learning","date":"2023-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"andrschl/cirl","path":"algs/cirl.py","file_url":"https://github.com/andrschl/cirl/blob/HEAD/algs/cirl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c6f978bd33eadfed","mcp_get_code":{"code_sha256":"c6f978bd33eadfed"}},{"arxiv_id":"2205.15372","paper":"/paper/optimistic-whittle-index-policy-online","title":"Optimistic Whittle Index Policy: Online Learning for Restless Bandits","date":"2022-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lily-x/online-rmab","path":"src/uc_whittle.py","file_url":"https://github.com/lily-x/online-rmab/blob/HEAD/src/uc_whittle.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c26495f6372f08b3","mcp_get_code":{"code_sha256":"c26495f6372f08b3"}},{"arxiv_id":"2204.11860","paper":"/paper/multi-objective-pointer-network-for","title":"Multi-objective Pointer Network for Combinatorial Optimization","date":"2022-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gaoly/mopn","path":"MOPN_type1_obj2/tasks/motsp.py","file_url":"https://github.com/gaoly/mopn/blob/HEAD/MOPN_type1_obj2/tasks/motsp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c284ef1374f699e8","mcp_get_code":{"code_sha256":"c284ef1374f699e8"}},{"arxiv_id":"2006.06051","paper":"/paper/learning-to-incentivize-other-learning-agents","title":"Learning to Incentivize Other Learning Agents","date":"2020-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"011235813/lio","path":"lio/alg/lio_ac.py","file_url":"https://github.com/011235813/lio/blob/HEAD/lio/alg/lio_ac.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fb4097e566da6fc0","mcp_get_code":{"code_sha256":"fb4097e566da6fc0"}},{"arxiv_id":"1805.07010","paper":"/paper/learning-permutations-with-sinkhorn-policy","title":"Learning Permutations with Sinkhorn Policy Gradient","date":"2018-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pemami4911/sinkhorn-policy-gradient.pytorch","path":"envs/mwm2D_task.py","file_url":"https://github.com/pemami4911/sinkhorn-policy-gradient.pytorch/blob/HEAD/envs/mwm2D_task.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"fbcc4ff1d7572f35","mcp_get_code":{"code_sha256":"fbcc4ff1d7572f35"}},{"arxiv_id":"1802.04240","paper":"/paper/reinforcement-learning-for-solving-the","title":"Reinforcement Learning for Solving the Vehicle Routing Problem","date":"2018-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ajayn1997/RL-VRP-PtrNtwrk","path":"Tasks/vrp.py","file_url":"https://github.com/ajayn1997/RL-VRP-PtrNtwrk/blob/HEAD/Tasks/vrp.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8c63c20097bd4bce","mcp_get_code":{"code_sha256":"8c63c20097bd4bce"}},{"arxiv_id":"1611.09940","paper":"/paper/neural-combinatorial-optimization-with","title":"Neural Combinatorial Optimization with Reinforcement Learning","date":"2016-11-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pemami4911/neural-combinatorial-rl-pytorch","path":"sorting_task.py","file_url":"https://github.com/pemami4911/neural-combinatorial-rl-pytorch/blob/HEAD/sorting_task.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"da119d24707d7943","mcp_get_code":{"code_sha256":"da119d24707d7943"}},{"arxiv_id":"1611.09940","paper":"/paper/neural-combinatorial-optimization-with","title":"Neural Combinatorial Optimization with Reinforcement Learning","date":"2016-11-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pemami4911/neural-combinatorial-rl-pytorch","path":"tsp_task.py","file_url":"https://github.com/pemami4911/neural-combinatorial-rl-pytorch/blob/HEAD/tsp_task.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"570a2842c2ff2ab8","mcp_get_code":{"code_sha256":"570a2842c2ff2ab8"}}]}