{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/layer-init","entry":"layer_init","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":46,"n_papers_ran":38,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":20,"n_samples_ran":12,"n_samples_fingerprinted":0,"n_places":51,"n_places_pointer_only":27,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":7,"ran_fixture":0,"ran":5,"unverified":8},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.21311","paper":"/paper/arxiv-2605-21311","title":"DeCoR: Design and Control Co-Optimization for Urban Streets Using Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"poudel-bibek/DeCoR","path":"ppo/models.py","file_url":"https://github.com/poudel-bibek/DeCoR/blob/HEAD/ppo/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d90a784ee56240f3","mcp_get_code":{"code_sha256":"d90a784ee56240f3"}},{"arxiv_id":"2603.00903","paper":"/paper/arxiv-2603-00903","title":"Principled Fast and Meta Knowledge Learners for Continual Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"datake/FAME","path":"Atari/models/dino_simple.py","file_url":"https://github.com/datake/FAME/blob/HEAD/Atari/models/dino_simple.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2603.00903","paper":"/paper/arxiv-2603-00903","title":"Principled Fast and Meta Knowledge Learners for Continual Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"datake/FAME","path":"Atari/models/FastMeta.py","file_url":"https://github.com/datake/FAME/blob/HEAD/Atari/models/FastMeta.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"2603.00903","paper":"/paper/arxiv-2603-00903","title":"Principled Fast and Meta Knowledge Learners for Continual Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"datake/FAME","path":"Atari/models/progressive_net.py","file_url":"https://github.com/datake/FAME/blob/HEAD/Atari/models/progressive_net.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cd674ceafb2bb134","mcp_get_code":{"code_sha256":"cd674ceafb2bb134"}},{"arxiv_id":"2602.13551","paper":"/paper/arxiv-2602-13551","title":"Small Reward Models via Backward Inference","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"yikee/FLIP","path":"open-instruct/open_instruct/reward_modeling.py","file_url":"https://github.com/yikee/FLIP/blob/HEAD/open-instruct/open_instruct/reward_modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"85c132fb4ad771ab","mcp_get_code":{"code_sha256":"85c132fb4ad771ab"}},{"arxiv_id":"2601.23075","paper":"/paper/arxiv-2601-23075","title":"RN-D: Discretized Categorical Actors for On-Policy Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"alwaysbyx/RND-RL","path":"rnd.py","file_url":"https://github.com/alwaysbyx/RND-RL/blob/HEAD/rnd.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f507fed062498260","mcp_get_code":{"code_sha256":"f507fed062498260"}},{"arxiv_id":"2601.21754","paper":"/paper/arxiv-2601-21754","title":"Language-based Trial and Error Falls Behind in the Era of Experience","date":"2026-01-29","month_inferred_from_arxiv_id":null,"title_source":"syntology","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2506.08417","paper":"/paper/offline-rl-with-smooth-ood-generalization-in","title":"Offline RL with Smooth OOD Generalization in Convex Hull and its Neighborhood","date":"2025-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yqpqry/sqog","path":"SQOG.py","file_url":"https://github.com/yqpqry/sqog/blob/HEAD/SQOG.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"47c37c5b075bacaf","mcp_get_code":{"code_sha256":"47c37c5b075bacaf"}},{"arxiv_id":"2504.06386","paper":null,"title":"arXiv:2504.06386","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"JacquesCloete/sport","path":"src/sport/rl/algos/projected_ppo/core.py","file_url":"https://github.com/JacquesCloete/sport/blob/HEAD/src/sport/rl/algos/projected_ppo/core.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ad8ba118b16e67f2","mcp_get_code":{"code_sha256":"ad8ba118b16e67f2"}},{"arxiv_id":"2503.23478","paper":"/paper/handling-delay-in-real-time-reinforcement","title":"Handling Delay in Real-Time Reinforcement Learning","date":"2025-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"2503.13538","paper":"/paper/from-demonstrations-to-rewards-alignment","title":"From Demonstrations to Rewards: Alignment Without Explicit Human Preferences","date":"2025-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Hong-Lab-UMN-ECE/IRLAlignment","path":"visualize_tokens.py","file_url":"https://github.com/Hong-Lab-UMN-ECE/IRLAlignment/blob/HEAD/visualize_tokens.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9ad5922df265477f","mcp_get_code":{"code_sha256":"9ad5922df265477f"}},{"arxiv_id":"2502.10550","paper":"/paper/memory-benchmark-robots-a-benchmark-for","title":"Memory, Benchmark & Robots: A Benchmark for Solving Complex Tasks with Reinforcement Learning","date":"2025-02-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CognitiveAISystems/MIKASA-Robo","path":"mikasa_robo_suite/rl/dataset_collectors/get_dataset_collectors_ckpt.py","file_url":"https://github.com/CognitiveAISystems/MIKASA-Robo/blob/HEAD/mikasa_robo_suite/rl/dataset_collectors/get_dataset_collectors_ckpt.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2502.08938","paper":"/paper/reevaluating-policy-gradient-methods-for","title":"Reevaluating Policy Gradient Methods for Imperfect-Information Games","date":"2025-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2502.08938","paper":"/paper/reevaluating-policy-gradient-methods-for","title":"Reevaluating Policy Gradient Methods for Imperfect-Information Games","date":"2025-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"2410.22728","paper":"/paper/offline-behavior-distillation","title":"Offline Behavior Distillation","date":"2024-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leaveslei/obd","path":"data_lib/syndset.py","file_url":"https://github.com/leaveslei/obd/blob/HEAD/data_lib/syndset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f297771ebb6401b8","mcp_get_code":{"code_sha256":"f297771ebb6401b8"}},{"arxiv_id":"2410.18870","paper":"/paper/end-to-end-training-for-recommendation-with","title":"End-to-end Training for Recommendation with Language-based User Profiles","date":"2024-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaolingao/langptune","path":"src/langptune_gemma.py","file_url":"https://github.com/zhaolingao/langptune/blob/HEAD/src/langptune_gemma.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9ad5922df265477f","mcp_get_code":{"code_sha256":"9ad5922df265477f"}},{"arxiv_id":"2410.10905","paper":"/paper/improving-generalization-on-the-procgen","title":"Improving Generalization on the ProcGen Benchmark with Simple Architectural Changes and Scale","date":"2024-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anndvision/vsop-3d","path":"vsop_3d/vsop_3d_procgen.py","file_url":"https://github.com/anndvision/vsop-3d/blob/HEAD/vsop_3d/vsop_3d_procgen.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d3b66680bcde8d15","mcp_get_code":{"code_sha256":"d3b66680bcde8d15"}},{"arxiv_id":"2410.08893","paper":"/paper/drama-mamba-enabled-model-based-reinforcement","title":"Drama: Mamba-Enabled Model-Based Reinforcement Learning Is Sample and Parameter Efficient","date":"2024-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"realwenlongwang/Drama","path":"tools.py","file_url":"https://github.com/realwenlongwang/Drama/blob/HEAD/tools.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3efc0819c7bc75e2","mcp_get_code":{"code_sha256":"3efc0819c7bc75e2"}},{"arxiv_id":"2407.03969","paper":"/paper/craftium-an-extensible-framework-for-creating","title":"Craftium: An Extensible Framework for Creating Reinforcement Learning Environments","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mikelma/craftium","path":"cleanrl_ppo_lstm_train.py","file_url":"https://github.com/mikelma/craftium/blob/HEAD/cleanrl_ppo_lstm_train.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2405.14226","paper":"/paper/variational-delayed-policy-optimization","title":"Variational Delayed Policy Optimization","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QingyuanWuNothing/VDPO","path":"nn.py","file_url":"https://github.com/QingyuanWuNothing/VDPO/blob/HEAD/nn.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2405.00662","paper":"/paper/no-representation-no-trust-connecting","title":"No Representation, No Trust: Connecting Representation, Collapse, and Trust Issues in PPO","date":"2024-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"claire-labo/no-representation-no-trust","path":"src/cleanrl/ppo_atari_original.py","file_url":"https://github.com/claire-labo/no-representation-no-trust/blob/HEAD/src/cleanrl/ppo_atari_original.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2405.00662","paper":"/paper/no-representation-no-trust-connecting","title":"No Representation, No Trust: Connecting Representation, Collapse, and Trust Issues in PPO","date":"2024-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"claire-labo/no-representation-no-trust","path":"src/cleanrl/ppo_atari_1model.py","file_url":"https://github.com/claire-labo/no-representation-no-trust/blob/HEAD/src/cleanrl/ppo_atari_1model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"064fcfc6277bdc44","mcp_get_code":{"code_sha256":"064fcfc6277bdc44"}},{"arxiv_id":"2404.16767","paper":"/paper/rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaolingao/rebel","path":"src/tldr/rebel.py","file_url":"https://github.com/zhaolingao/rebel/blob/HEAD/src/tldr/rebel.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9ad5922df265477f","mcp_get_code":{"code_sha256":"9ad5922df265477f"}},{"arxiv_id":"2404.10728","paper":"/paper/randomized-exploration-in-cooperative-multi","title":"Randomized Exploration in Cooperative Multi-Agent Reinforcement Learning","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"panxulab/MARL-CoopTS","path":"components/network.py","file_url":"https://github.com/panxulab/MARL-CoopTS/blob/HEAD/components/network.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3a9ab37bf0498c42","mcp_get_code":{"code_sha256":"3a9ab37bf0498c42"}},{"arxiv_id":"2404.08495","paper":"/paper/dataset-reset-policy-optimization-for-rlhf","title":"Dataset Reset Policy Optimization for RLHF","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"9ad5922df265477f","mcp_get_code":{"code_sha256":"9ad5922df265477f"}},{"arxiv_id":"2404.07099","paper":"/paper/rethinking-out-of-distribution-detection-for","title":"Rethinking Out-of-Distribution Detection for Reinforcement Learning: Advancing Methods for Evaluation and Detection","date":"2024-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"linasnas/dexter","path":"src/train_test_detector_discrete_env.py","file_url":"https://github.com/linasnas/dexter/blob/HEAD/src/train_test_detector_discrete_env.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"244a24e0c333afbe","mcp_get_code":{"code_sha256":"244a24e0c333afbe"}},{"arxiv_id":"2403.17031","paper":"/paper/the-n-implementation-details-of-rlhf-with-ppo","title":"The N+ Implementation Details of RLHF with PPO: A Case Study on TL;DR Summarization","date":"2024-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vwxyzjn/summarize_from_feedback_details","path":"summarize_from_feedback_details/ppo.py","file_url":"https://github.com/vwxyzjn/summarize_from_feedback_details/blob/HEAD/summarize_from_feedback_details/ppo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9ad5922df265477f","mcp_get_code":{"code_sha256":"9ad5922df265477f"}},{"arxiv_id":"2402.03903","paper":"/paper/compound-returns-reduce-variance-in","title":"Averaging $n$-step Returns Reduces Variance in Reinforcement Learning","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"brett-daley/averaging-nstep-returns","path":"run_ppo.py","file_url":"https://github.com/brett-daley/averaging-nstep-returns/blob/HEAD/run_ppo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2402.03141","paper":"/paper/boosting-long-delayed-reinforcement-learning","title":"Boosting Reinforcement Learning with Strongly Delayed Feedback Through Auxiliary Short Delays","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QingyuanWuNothing/AD-RL","path":"nn.py","file_url":"https://github.com/QingyuanWuNothing/AD-RL/blob/HEAD/nn.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2401.14151","paper":"/paper/true-knowledge-comes-from-practice-aligning","title":"True Knowledge Comes from Practice: Aligning LLMs with Embodied Environments via Reinforcement Learning","date":"2024-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weihaotan/twosome","path":"twosome/overcooked/ppo_llm_pomdp.py","file_url":"https://github.com/weihaotan/twosome/blob/HEAD/twosome/overcooked/ppo_llm_pomdp.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2401.14151","paper":"/paper/true-knowledge-comes-from-practice-aligning","title":"True Knowledge Comes from Practice: Aligning LLMs with Embodied Environments via Reinforcement Learning","date":"2024-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weihaotan/twosome","path":"twosome/overcooked/policy_pomdp.py","file_url":"https://github.com/weihaotan/twosome/blob/HEAD/twosome/overcooked/policy_pomdp.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"2312.10812","paper":"/paper/learning-to-act-without-actions","title":"Learning to Act without Actions","date":"2023-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"schmidtdominik/LAPO","path":"lapo/models.py","file_url":"https://github.com/schmidtdominik/LAPO/blob/HEAD/lapo/models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"79a9507515948442","mcp_get_code":{"code_sha256":"79a9507515948442"}},{"arxiv_id":"2310.17805","paper":"/paper/reward-scale-robustness-for-proximal-policy","title":"Reward Scale Robustness for Proximal Policy Optimization via DreamerV3 Tricks","date":"2023-10-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RyanNavillus/PPO-v3","path":"ppo_v3/ppo_atari_envpool_resnet.py","file_url":"https://github.com/RyanNavillus/PPO-v3/blob/HEAD/ppo_v3/ppo_atari_envpool_resnet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2309.14597","paper":"/paper/policy-optimization-in-a-noisy-neighborhood-1","title":"Policy Optimization in a Noisy Neighborhood: On Return Landscapes in Continuous Control","date":"2023-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nathanrahn/return-landscapes","path":"run_algorithm/experiments/ppo_atari_envpool.py","file_url":"https://github.com/nathanrahn/return-landscapes/blob/HEAD/run_algorithm/experiments/ppo_atari_envpool.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2308.07795","paper":"/paper/learning-to-identify-critical-states-for","title":"Learning to Identify Critical States for Reinforcement Learning from Videos","date":"2023-08-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ai-initiative-kaust/videorlcs","path":"Policy_Improvement/Atari/atari_network.py","file_url":"https://github.com/ai-initiative-kaust/videorlcs/blob/HEAD/Policy_Improvement/Atari/atari_network.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"372e9511cee496ef","mcp_get_code":{"code_sha256":"372e9511cee496ef"}},{"arxiv_id":"2305.11476","paper":"/paper/learning-diverse-risk-preferences-in","title":"Learning Diverse Risk Preferences in Population-based Self-play","date":"2023-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2303.04651","paper":"/paper/mcts-geb-monte-carlo-tree-search-is-a-good-e","title":"MCTS-GEB: Monte Carlo Tree Search is a Good E-graph Builder","date":"2023-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ucamrl/eqs","path":"omelette-original/rejoice/networks.py","file_url":"https://github.com/ucamrl/eqs/blob/HEAD/omelette-original/rejoice/networks.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"2212.07536","paper":"/paper/robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2210.03022","paper":"/paper/stateful-active-facilitator-coordination-and","title":"Stateful active facilitator: Coordination and Environmental Heterogeneity in Cooperative Multi-Agent Reinforcement Learning","date":"2022-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jaggbow/saf","path":"src/policies/saf.py","file_url":"https://github.com/jaggbow/saf/blob/HEAD/src/policies/saf.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2206.10558","paper":"/paper/envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vwxyzjn/envpool-cleanrl","path":"ppo_atari_envpool.py","file_url":"https://github.com/vwxyzjn/envpool-cleanrl/blob/HEAD/ppo_atari_envpool.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2111.08819","paper":"/paper/cleanrl-high-quality-single-file","title":"CleanRL: High-quality Single-file Implementations of Deep Reinforcement Learning Algorithms","date":"2021-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2111.02997","paper":"/paper/global-optimality-and-finite-sample-analysis","title":"Global Optimality and Finite Sample Analysis of Softmax Off-Policy Actor Critic under State Distribution Mismatch","date":"2021-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShangtongZhang/DeepRL","path":"deep_rl/network/network_utils.py","file_url":"https://github.com/ShangtongZhang/DeepRL/blob/HEAD/deep_rl/network/network_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e47770c24cec6408","mcp_get_code":{"code_sha256":"e47770c24cec6408"}},{"arxiv_id":"2106.07841","paper":"/paper/randomized-exploration-for-reinforcement","title":"Randomized Exploration for Reinforcement Learning with General Value Function Approximation","date":"2021-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qlan3/Explorer","path":"components/network.py","file_url":"https://github.com/qlan3/Explorer/blob/HEAD/components/network.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3a9ab37bf0498c42","mcp_get_code":{"code_sha256":"3a9ab37bf0498c42"}},{"arxiv_id":"1906.09323","paper":"/paper/reinforcement-learning-with-convex","title":"Reinforcement Learning with Convex Constraints","date":"2019-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xkianteb/ApproPO","path":"ApproPO/nets.py","file_url":"https://github.com/xkianteb/ApproPO/blob/HEAD/ApproPO/nets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e47770c24cec6408","mcp_get_code":{"code_sha256":"e47770c24cec6408"}},{"arxiv_id":"1805.09801","paper":"/paper/meta-gradient-reinforcement-learning","title":"Meta-Gradient Reinforcement Learning","date":"2018-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RobvanGastel/meta-rl-algorithms","path":"utils/misc.py","file_url":"https://github.com/RobvanGastel/meta-rl-algorithms/blob/HEAD/utils/misc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e12d84ba1f73ff01","mcp_get_code":{"code_sha256":"e12d84ba1f73ff01"}},{"arxiv_id":"1703.03400","paper":"/paper/model-agnostic-meta-learning-for-fast","title":"Model-Agnostic Meta-Learning for Fast Adaptation of Deep Networks","date":"2017-03-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Octavio-Pappalardo/MAML_for_RL_pytorch","path":"FO-MAML_distributed/network.py","file_url":"https://github.com/Octavio-Pappalardo/MAML_for_RL_pytorch/blob/HEAD/FO-MAML_distributed/network.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"1611.05763","paper":"/paper/learning-to-reinforcement-learn","title":"Learning to reinforcement learn","date":"2016-11-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"1611.02779","paper":"/paper/rl2-fast-reinforcement-learning-via-slow","title":"RL$^2$: Fast Reinforcement Learning via Slow Reinforcement Learning","date":"2016-11-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Octavio-Pappalardo/RL2-implementation-pytorch","path":"agent_and_tbpttPPO.py","file_url":"https://github.com/Octavio-Pappalardo/RL2-implementation-pytorch/blob/HEAD/agent_and_tbpttPPO.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"openreview_KBt9iIfQAi","paper":null,"title":"arXiv:openreview_KBt9iIfQAi","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yanmluk/causal-xrl","path":"cxrl_microRTS/lib/gym_microrts_rl/env_tools.py","file_url":"https://github.com/yanmluk/causal-xrl/blob/HEAD/cxrl_microRTS/lib/gym_microrts_rl/env_tools.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e01d3118f70d5b05","mcp_get_code":{"code_sha256":"e01d3118f70d5b05"}},{"arxiv_id":"aaai_29188","paper":null,"title":"arXiv:aaai_29188","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Jackory/RPBT","path":"toyexample/rppo.py","file_url":"https://github.com/Jackory/RPBT/blob/HEAD/toyexample/rppo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"beb42d440ccf2925","mcp_get_code":{"code_sha256":"beb42d440ccf2925"}},{"arxiv_id":"2024.acl-long.729","paper":null,"title":"arXiv:2024.acl-long.729","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"xjw-nlp/SimCAS","path":"modeling_bart_ours.py","file_url":"https://github.com/xjw-nlp/SimCAS/blob/HEAD/modeling_bart_ours.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3cd6d9bbebd8f0fe","mcp_get_code":{"code_sha256":"3cd6d9bbebd8f0fe"}}]}