{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/value","entry":"Value","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":10,"n_papers_ran":8,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":12,"n_samples_ran":10,"n_samples_fingerprinted":4,"n_places":12,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":10,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.01205","paper":"/paper/arxiv-2608-01205","title":"ReBRAC-v2: The Return of the King","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"JongseongChae/FAC","path":"agents/rebrac.py","file_url":"https://github.com/JongseongChae/FAC/blob/HEAD/agents/rebrac.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"49c47a75385843a1","mcp_get_code":{"code_sha256":"49c47a75385843a1"}},{"arxiv_id":"2608.01205","paper":"/paper/arxiv-2608-01205","title":"ReBRAC-v2: The Return of the King","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Simple-Robotics/guided-flow-policy","path":"agents/rebrac.py","file_url":"https://github.com/Simple-Robotics/guided-flow-policy/blob/HEAD/agents/rebrac.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"54aa459fb400c5f8","mcp_get_code":{"code_sha256":"54aa459fb400c5f8"}},{"arxiv_id":"2605.11711","paper":"/paper/arxiv-2605-11711","title":"Debiased Model-based Representations for Sample-efficient Continuous Control","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"dmksjfl/DR.Q","path":"DRQ/DRQ.py","file_url":"https://github.com/dmksjfl/DR.Q/blob/HEAD/DRQ/DRQ.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ff2bd29a65372428","mcp_get_code":{"code_sha256":"ff2bd29a65372428"}},{"arxiv_id":"2605.01663","paper":"/paper/arxiv-2605-01663","title":"Towards Efficient and Expressive Offline RL via Flow-Anchored Noise-conditioned Q-Learning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"brianlsy98/FAN","path":"agents/fan.py","file_url":"https://github.com/brianlsy98/FAN/blob/HEAD/agents/fan.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9235803e8ce9944a","mcp_get_code":{"code_sha256":"9235803e8ce9944a"}},{"arxiv_id":"2603.14245","paper":"/paper/arxiv-2603-14245","title":"GoldenStart: Q-Guided Priors and Entropy Control for Distilling Flow Policies","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"ZhHe11/GSFlow-RL","path":"agents/gsflow.py","file_url":"https://github.com/ZhHe11/GSFlow-RL/blob/HEAD/agents/gsflow.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"131d043e72cb8f14","mcp_get_code":{"code_sha256":"131d043e72cb8f14"}},{"arxiv_id":"2601.21306","paper":"/paper/arxiv-2601-21306","title":"The Surprising Difficulty of Search in Model-Based Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"facebookresearch/MRSQ","path":"MRSQ/MRSQ.py","file_url":"https://github.com/facebookresearch/MRSQ/blob/HEAD/MRSQ/MRSQ.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"5c4890147136a329","mcp_get_code":{"code_sha256":"5c4890147136a329"}},{"arxiv_id":"2506.08902","paper":"/paper/intention-conditioned-flow-occupancy-models","title":"Intention-Conditioned Flow Occupancy Models","date":"2025-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chongyi-zheng/infom","path":"agents/infom.py","file_url":"https://github.com/chongyi-zheng/infom/blob/HEAD/agents/infom.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"028beb23076e7758","mcp_get_code":{"code_sha256":"028beb23076e7758"}},{"arxiv_id":"2405.19909","paper":"/paper/adaptive-advantage-guided-policy","title":"Adaptive Advantage-Guided Policy Regularization for Offline Reinforcement Learning","date":"2024-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltlhuuu/a2pr","path":"A2PR.py","file_url":"https://github.com/ltlhuuu/a2pr/blob/HEAD/A2PR.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f135619af421a978","mcp_get_code":{"code_sha256":"f135619af421a978"}},{"arxiv_id":"2210.03078","paper":"/paper/rainier-reinforced-knowledge-introspector-for","title":"Rainier: Reinforced Knowledge Introspector for Commonsense Question Answering","date":"2022-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liujch1998/rainier","path":"rainier/ppo.py","file_url":"https://github.com/liujch1998/rainier/blob/HEAD/rainier/ppo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d77834c9fb30a1e7","mcp_get_code":{"code_sha256":"d77834c9fb30a1e7"}},{"arxiv_id":"2110.06169","paper":"/paper/offline-reinforcement-learning-with-implicit","title":"Offline Reinforcement Learning with Implicit Q-Learning","date":"2021-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"orrivlin/implicit-q-learning","path":"IQL.py","file_url":"https://github.com/orrivlin/implicit-q-learning/blob/HEAD/IQL.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ab256778314dd6eb","mcp_get_code":{"code_sha256":"ab256778314dd6eb"}},{"arxiv_id":"2110.06169","paper":"/paper/offline-reinforcement-learning-with-implicit","title":"Offline Reinforcement Learning with Implicit Q-Learning","date":"2021-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BY571/Implicit-Q-Learning","path":"agent.py","file_url":"https://github.com/BY571/Implicit-Q-Learning/blob/HEAD/agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f90b550029795ff8","mcp_get_code":{"code_sha256":"f90b550029795ff8"}},{"arxiv_id":"1801.01290","paper":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning","path":"src/agents.py","file_url":"https://github.com/MatthieuSarkis/Portfolio-Optimization-and-Goal-Based-Investment-with-Reinforcement-Learning/blob/HEAD/src/agents.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f8845c35735f20bc","mcp_get_code":{"code_sha256":"f8845c35735f20bc"}}]}