{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/make-env","entry":"make_env","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":50,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":51,"n_samples_ran":8,"n_samples_fingerprinted":0,"n_places":56,"n_places_pointer_only":17,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":1,"ran":7,"unverified":43},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.01761","paper":"/paper/arxiv-2609-01761","title":"Pooling and Drift in Delayed Bandits","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"melikabaghi/state-exp3","path":"code/algo.py","file_url":"https://github.com/melikabaghi/state-exp3/blob/HEAD/code/algo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3b178984945a0cec","mcp_get_code":{"code_sha256":"3b178984945a0cec"}},{"arxiv_id":"2608.02332","paper":"/paper/arxiv-2608-02332","title":"Diffusion Policy with Behavioral Advantage Correction for Offline Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"sfujim/BCQ","path":"discrete_BCQ/utils.py","file_url":"https://github.com/sfujim/BCQ/blob/HEAD/discrete_BCQ/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4410a954fa46ebfb","mcp_get_code":{"code_sha256":"4410a954fa46ebfb"}},{"arxiv_id":"2607.29617","paper":"/paper/arxiv-2607-29617","title":"When Does On-Policy Interaction Help? Representational Tradeoffs in Value-Based Imitation Learning","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"lviano/ovi","path":"gym_code/run_traj_experiment.py","file_url":"https://github.com/lviano/ovi/blob/HEAD/gym_code/run_traj_experiment.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4e536803a3f0eeb0","mcp_get_code":{"code_sha256":"4e536803a3f0eeb0"}},{"arxiv_id":"2607.22832","paper":"/paper/arxiv-2607-22832","title":"MEMENTO: Memory-Guided Memetic Code-as-Policy Evolution","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"sygkounas/MEMENTO","path":"MEMENTO/inference.py","file_url":"https://github.com/sygkounas/MEMENTO/blob/HEAD/MEMENTO/inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bef19eafa0358309","mcp_get_code":{"code_sha256":"bef19eafa0358309"}},{"arxiv_id":"2510.15382","paper":"/paper/arxiv-2510-15382","title":"Towards Robust Zero-Shot Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"seohongpark/HILP","path":"hilp_gcrl/src/d4rl_utils.py","file_url":"https://github.com/seohongpark/HILP/blob/HEAD/hilp_gcrl/src/d4rl_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b3d75720ab55c2c2","mcp_get_code":{"code_sha256":"b3d75720ab55c2c2"}},{"arxiv_id":"2508.07626","paper":"/paper/arxiv-2508-07626","title":"AR-VRM: Imitating Human Motions for Visual Robot Manipulation with Analogical Reasoning","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"idejie/ar","path":"evaluate.py","file_url":"https://github.com/idejie/ar/blob/HEAD/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"21fb46132c505c9e","mcp_get_code":{"code_sha256":"21fb46132c505c9e"}},{"arxiv_id":"2505.05262","paper":"/paper/enhancing-cooperative-multi-agent","title":"Enhancing Cooperative Multi-Agent Reinforcement Learning with State Modelling and Adversarial Exploration","date":"2025-05-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"semitable/multiagent-particle-envs","path":"make_env.py","file_url":"https://github.com/semitable/multiagent-particle-envs/blob/HEAD/make_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"e0db348c4be68fc5","mcp_get_code":{"code_sha256":"e0db348c4be68fc5"}},{"arxiv_id":"2501.11039","paper":"/paper/beyond-any-shot-adaptation-predicting","title":"Beyond Any-Shot Adaptation: Predicting Optimization Outcome for Robustness Gains without Extra Pay","date":"2025-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-rllab/mpts","path":"MetaRL/sampler.py","file_url":"https://github.com/thu-rllab/mpts/blob/HEAD/MetaRL/sampler.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4246e3602f23a27a","mcp_get_code":{"code_sha256":"4246e3602f23a27a"}},{"arxiv_id":"2408.17355","paper":"/paper/bidirectional-decoding-improving-action","title":"Bidirectional Decoding: Improving Action Chunking via Guided Test-Time Sampling","date":"2024-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jubayer-hamid/bid_lerobot","path":"lerobot/common/envs/factory.py","file_url":"https://github.com/jubayer-hamid/bid_lerobot/blob/HEAD/lerobot/common/envs/factory.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a46ac2680ddd2dbe","mcp_get_code":{"code_sha256":"a46ac2680ddd2dbe"}},{"arxiv_id":"2407.03969","paper":"/paper/craftium-an-extensible-framework-for-creating","title":"Craftium: An Extensible Framework for Creating Reinforcement Learning Environments","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mikelma/craftium","path":"cleanrl_ppo_lstm_train.py","file_url":"https://github.com/mikelma/craftium/blob/HEAD/cleanrl_ppo_lstm_train.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"47d8fe4d97b14b66","mcp_get_code":{"code_sha256":"47d8fe4d97b14b66"}},{"arxiv_id":"2407.03969","paper":"/paper/craftium-an-extensible-framework-for-creating","title":"Craftium: An Extensible Framework for Creating Reinforcement Learning Environments","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mikelma/craftium","path":"cleanrl_ppo_train.py","file_url":"https://github.com/mikelma/craftium/blob/HEAD/cleanrl_ppo_train.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"362e38152eaf2597","mcp_get_code":{"code_sha256":"362e38152eaf2597"}},{"arxiv_id":"2405.00662","paper":"/paper/no-representation-no-trust-connecting","title":"No Representation, No Trust: Connecting Representation, Collapse, and Trust Issues in PPO","date":"2024-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"claire-labo/no-representation-no-trust","path":"src/cleanrl/ppo_mujoco_original.py","file_url":"https://github.com/claire-labo/no-representation-no-trust/blob/HEAD/src/cleanrl/ppo_mujoco_original.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"06ab2be7a18cea95","mcp_get_code":{"code_sha256":"06ab2be7a18cea95"}},{"arxiv_id":"2405.00662","paper":"/paper/no-representation-no-trust-connecting","title":"No Representation, No Trust: Connecting Representation, Collapse, and Trust Issues in PPO","date":"2024-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"claire-labo/no-representation-no-trust","path":"src/cleanrl/ppo_mujoco_torchrl.py","file_url":"https://github.com/claire-labo/no-representation-no-trust/blob/HEAD/src/cleanrl/ppo_mujoco_torchrl.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6bdd010a8786120c","mcp_get_code":{"code_sha256":"6bdd010a8786120c"}},{"arxiv_id":"2402.15567","paper":"/paper/foundation-policies-with-hilbert","title":"Foundation Policies with Hilbert Representations","date":"2024-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seohongpark/hilp","path":"hilp_gcrl/src/d4rl_utils.py","file_url":"https://github.com/seohongpark/hilp/blob/HEAD/hilp_gcrl/src/d4rl_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b3d75720ab55c2c2","mcp_get_code":{"code_sha256":"b3d75720ab55c2c2"}},{"arxiv_id":"2402.10228","paper":"/paper/hyperagent-a-simple-scalable-efficient-and","title":"Q-Star Meets Scalable Posterior Sampling: Bridging Theory and Practice via HyperAgent","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"szrlee/hyperagent","path":"hyperagent/env/utils.py","file_url":"https://github.com/szrlee/hyperagent/blob/HEAD/hyperagent/env/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7987a0deae85f58b","mcp_get_code":{"code_sha256":"7987a0deae85f58b"}},{"arxiv_id":"2402.03903","paper":"/paper/compound-returns-reduce-variance-in","title":"Averaging $n$-step Returns Reduces Variance in Reinforcement Learning","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"brett-daley/averaging-nstep-returns","path":"run_ppo.py","file_url":"https://github.com/brett-daley/averaging-nstep-returns/blob/HEAD/run_ppo.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"06ab2be7a18cea95","mcp_get_code":{"code_sha256":"06ab2be7a18cea95"}},{"arxiv_id":"2401.14151","paper":"/paper/true-knowledge-comes-from-practice-aligning","title":"True Knowledge Comes from Practice: Aligning LLMs with Embodied Environments via Reinforcement Learning","date":"2024-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weihaotan/twosome","path":"twosome/virtualhome/inference_ppo_llm_v1.py","file_url":"https://github.com/weihaotan/twosome/blob/HEAD/twosome/virtualhome/inference_ppo_llm_v1.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"2e54f2ecd0d61798","mcp_get_code":{"code_sha256":"2e54f2ecd0d61798"}},{"arxiv_id":"2312.13139","paper":"/paper/unleashing-large-scale-video-generative-pre","title":"Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation","date":"2023-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/gr-1","path":"evaluate_calvin.py","file_url":"https://github.com/bytedance/gr-1/blob/HEAD/evaluate_calvin.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8e37b3c3fbc7c4a5","mcp_get_code":{"code_sha256":"8e37b3c3fbc7c4a5"}},{"arxiv_id":"2306.00036","paper":"/paper/symmetry-aware-robot-design-with-structured","title":"Symmetry-Aware Robot Design with Structured Subgroups","date":"2023-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"drdh/sard","path":"design_opt/derl_terrain_builder.py","file_url":"https://github.com/drdh/sard/blob/HEAD/design_opt/derl_terrain_builder.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"717d786b79662910","mcp_get_code":{"code_sha256":"717d786b79662910"}},{"arxiv_id":"2301.11741","paper":"/paper/outcome-directed-reinforcement-learning-by","title":"Outcome-directed Reinforcement Learning by Uncertainty & Temporal Distance-Aware Curriculum Goal Generation","date":"2023-01-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Stilwell-Git/Hindsight-Goal-Generation","path":"learner/hgg.py","file_url":"https://github.com/Stilwell-Git/Hindsight-Goal-Generation/blob/HEAD/learner/hgg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d9e1da7c33277df3","mcp_get_code":{"code_sha256":"d9e1da7c33277df3"}},{"arxiv_id":"2212.05698","paper":"/paper/modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/modem","path":"env.py","file_url":"https://github.com/facebookresearch/modem/blob/HEAD/env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"88d563b866d0550c","mcp_get_code":{"code_sha256":"88d563b866d0550c"}},{"arxiv_id":"2212.02705","paper":"/paper/what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"susanbao/rmarl_code","path":"rmarl/experiments/train_with_perturbed_network.py","file_url":"https://github.com/susanbao/rmarl_code/blob/HEAD/rmarl/experiments/train_with_perturbed_network.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"24a950999549b85d","mcp_get_code":{"code_sha256":"24a950999549b85d"}},{"arxiv_id":"2212.02705","paper":"/paper/what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"susanbao/rmarl_code","path":"multiagent-particle-envs/make_env.py","file_url":"https://github.com/susanbao/rmarl_code/blob/HEAD/multiagent-particle-envs/make_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"a87ea3074224c0e8","mcp_get_code":{"code_sha256":"a87ea3074224c0e8"}},{"arxiv_id":"2211.10282","paper":"/paper/exploring-through-random-curiosity-with","title":"Exploring through Random Curiosity with General Value Functions","date":"2022-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aditya-ramesh-10/exploring-through-rcgvf","path":"mgrid_utils/env.py","file_url":"https://github.com/aditya-ramesh-10/exploring-through-rcgvf/blob/HEAD/mgrid_utils/env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2f470de73c0eda3c","mcp_get_code":{"code_sha256":"2f470de73c0eda3c"}},{"arxiv_id":"2210.03209","paper":"/paper/self-adaptive-driving-in-nonstationary","title":"Self-Adaptive Driving in Nonstationary Environments through Conjectural Online Lookahead Adaptation","date":null,"month_inferred_from_arxiv_id":"2022-10","title_source":"archive","repo":"panshark/cola","path":"environment/atari.py","file_url":"https://github.com/panshark/cola/blob/HEAD/environment/atari.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d4af84ca3df6b55e","mcp_get_code":{"code_sha256":"d4af84ca3df6b55e"}},{"arxiv_id":"2210.03209","paper":"/paper/self-adaptive-driving-in-nonstationary","title":"Self-Adaptive Driving in Nonstationary Environments through Conjectural Online Lookahead Adaptation","date":null,"month_inferred_from_arxiv_id":"2022-10","title_source":"archive","repo":"panshark/cola","path":"environment/utils.py","file_url":"https://github.com/panshark/cola/blob/HEAD/environment/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"39db26d3d8bcbd5e","mcp_get_code":{"code_sha256":"39db26d3d8bcbd5e"}},{"arxiv_id":"2207.08894","paper":"/paper/a-deep-reinforcement-learning-approach-for-13","title":"A Deep Reinforcement Learning Approach for Finding Non-Exploitable Strategies in Two-Player Atari Games","date":"2022-07-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"quantumiracle/mars","path":"mars/env/import_env.py","file_url":"https://github.com/quantumiracle/mars/blob/HEAD/mars/env/import_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7bd11d0025602c10","mcp_get_code":{"code_sha256":"7bd11d0025602c10"}},{"arxiv_id":"2206.14349","paper":"/paper/fleet-dagger-interactive-robot-fleet-learning","title":"Fleet-DAgger: Interactive Robot Fleet Learning with Scalable Human Supervision","date":"2022-06-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"berkeleyautomation/ifl_benchmark","path":"env/make_utils.py","file_url":"https://github.com/berkeleyautomation/ifl_benchmark/blob/HEAD/env/make_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"18c7c52ef841c0d4","mcp_get_code":{"code_sha256":"18c7c52ef841c0d4"}},{"arxiv_id":"2206.10558","paper":"/paper/envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vwxyzjn/envpool-cleanrl","path":"ppo_continuous_action.py","file_url":"https://github.com/vwxyzjn/envpool-cleanrl/blob/HEAD/ppo_continuous_action.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3ee033959764bfd6","mcp_get_code":{"code_sha256":"3ee033959764bfd6"}},{"arxiv_id":"2204.02877","paper":"/paper/pandr-fast-adaptation-to-new-environments","title":"PAnDR: Fast Adaptation to New Environments from Offline Experiences via Decoupling Policy and Environment Representations","date":"2022-04-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tristandeleu/pytorch-maml-rl","path":"maml_rl/samplers/sampler.py","file_url":"https://github.com/tristandeleu/pytorch-maml-rl/blob/HEAD/maml_rl/samplers/sampler.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c2e1b382dce1e26b","mcp_get_code":{"code_sha256":"c2e1b382dce1e26b"}},{"arxiv_id":"2111.00134","paper":"/paper/context-meta-reinforcement-learning-via","title":"Context Meta-Reinforcement Learning via Neuromodulation","date":"2021-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dlpbc/nm-metarl","path":"nm_cavia/rl/sampler.py","file_url":"https://github.com/dlpbc/nm-metarl/blob/HEAD/nm_cavia/rl/sampler.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"117528af3ffaa8d8","mcp_get_code":{"code_sha256":"117528af3ffaa8d8"}},{"arxiv_id":"2109.05940","paper":"/paper/cross-domain-robot-imitation-with-invariant","title":"Cross Domain Robot Imitation with Invariant Representation","date":"2021-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaohengyin/irgail_example","path":"imitation_learning/env.py","file_url":"https://github.com/zhaohengyin/irgail_example/blob/HEAD/imitation_learning/env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e10629e58933d321","mcp_get_code":{"code_sha256":"e10629e58933d321"}},{"arxiv_id":"2103.14274","paper":"/paper/character-controllers-using-motion-vaes","title":"Character Controllers Using Motion VAEs","date":"2021-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"electronicarts/character-motion-vaes","path":"common/envs_utils.py","file_url":"https://github.com/electronicarts/character-motion-vaes/blob/HEAD/common/envs_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"3a9ecb11281ffe04","mcp_get_code":{"code_sha256":"3a9ecb11281ffe04"}},{"arxiv_id":"2010.09776","paper":"/paper/smarts-scalable-multi-agent-reinforcement","title":"SMARTS: Scalable Multi-Agent Reinforcement Learning Training School for Autonomous Driving","date":"2020-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mcederle99/MAD4QN-PS","path":"util_rgb.py","file_url":"https://github.com/mcederle99/MAD4QN-PS/blob/HEAD/util_rgb.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"4e65a552c24a864d","mcp_get_code":{"code_sha256":"4e65a552c24a864d"}},{"arxiv_id":"2010.09635","paper":"/paper/deep-reinforcement-learning-with-population","title":"Deep Reinforcement Learning with Population-Coded Spiking Neural Network for Continuous Control","date":"2020-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"combra-lab/pop-spiking-deep-rl","path":"popsan_drl/popsan_ppo/ppo_cuda_norm.py","file_url":"https://github.com/combra-lab/pop-spiking-deep-rl/blob/HEAD/popsan_drl/popsan_ppo/ppo_cuda_norm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c2985d7efcd097a6","mcp_get_code":{"code_sha256":"c2985d7efcd097a6"}},{"arxiv_id":"2010.01062","paper":"/paper/exploration-in-approximate-hyper-state-space","title":"Exploration in Approximate Hyper-State Space for Meta Reinforcement Learning","date":"2020-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lmzintgraf/hyperx","path":"vae.py","file_url":"https://github.com/lmzintgraf/hyperx/blob/HEAD/vae.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"444582a5af765854","mcp_get_code":{"code_sha256":"444582a5af765854"}},{"arxiv_id":"2007.06049","paper":"/paper/an-equivalence-between-loss-functions-and-non","title":"An Equivalence between Loss Functions and Non-Uniform Sampling in Experience Replay","date":"2020-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sfujim/LAP-PAL","path":"discrete/utils.py","file_url":"https://github.com/sfujim/LAP-PAL/blob/HEAD/discrete/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4410a954fa46ebfb","mcp_get_code":{"code_sha256":"4410a954fa46ebfb"}},{"arxiv_id":"2006.13760","paper":"/paper/the-nethack-learning-environment","title":"The NetHack Learning Environment","date":"2020-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Pieter-Cawood/Reinforcement-Learning","path":"NLE_DQN/Agent.py","file_url":"https://github.com/Pieter-Cawood/Reinforcement-Learning/blob/HEAD/NLE_DQN/Agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2f6ab6f543c54d76","mcp_get_code":{"code_sha256":"2f6ab6f543c54d76"}},{"arxiv_id":"2003.07305","paper":"/paper/discor-corrective-feedback-in-reinforcement","title":"DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","date":"2020-03-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"toshikwa/discor.pytorch","path":"discor/env.py","file_url":"https://github.com/toshikwa/discor.pytorch/blob/HEAD/discor/env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6b8bc5549d0da3fa","mcp_get_code":{"code_sha256":"6b8bc5549d0da3fa"}},{"arxiv_id":"1911.02140","paper":"/paper/fully-parameterized-quantile-function-for","title":"Fully Parameterized Quantile Function for Distributional Reinforcement Learning","date":"2019-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BY571/FQF-and-Extensions","path":"wrapper.py","file_url":"https://github.com/BY571/FQF-and-Extensions/blob/HEAD/wrapper.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"06fc02646fefa30a","mcp_get_code":{"code_sha256":"06fc02646fefa30a"}},{"arxiv_id":"1910.12154","paper":"/paper/zpd-teaching-strategies-for-deep","title":"ZPD Teaching Strategies for Deep Reinforcement Learning from Demonstrations","date":"2019-10-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RevanMacQueen/LearningFromHumans","path":"lfh/envs/atari.py","file_url":"https://github.com/RevanMacQueen/LearningFromHumans/blob/HEAD/lfh/envs/atari.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"06128276029b84db","mcp_get_code":{"code_sha256":"06128276029b84db"}},{"arxiv_id":"1910.04700","paper":"/paper/assistive-gym-a-physics-simulation-framework","title":"Assistive Gym: A Physics Simulation Framework for Assistive Robotics","date":"2019-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Healthcare-Robotics/assistive-gym","path":"assistive_gym/learn.py","file_url":"https://github.com/Healthcare-Robotics/assistive-gym/blob/HEAD/assistive_gym/learn.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4871fcb1309a2d93","mcp_get_code":{"code_sha256":"4871fcb1309a2d93"}},{"arxiv_id":"1906.01202","paper":"/paper/learning-transferable-cooperative-behavior-in","title":"Learning Transferable Cooperative Behavior in Multi-Agent Teams","date":"2019-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sumitsk/matrl","path":"mape/make_env.py","file_url":"https://github.com/sumitsk/matrl/blob/HEAD/mape/make_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e90be820b06b21f1","mcp_get_code":{"code_sha256":"e90be820b06b21f1"}},{"arxiv_id":"1905.10615","paper":"/paper/adversarial-policies-attacking-deep","title":"Adversarial Policies: Attacking Deep Reinforcement Learning","date":"2019-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HumanCompatibleAI/adversarial-policies","path":"experiments/planning/common.py","file_url":"https://github.com/HumanCompatibleAI/adversarial-policies/blob/HEAD/experiments/planning/common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"821d38fa42efe55c","mcp_get_code":{"code_sha256":"821d38fa42efe55c"}},{"arxiv_id":"1905.05408","paper":"/paper/qtran-learning-to-factorize-with","title":"QTRAN: Learning to Factorize with Transformation for Cooperative Multi-Agent Reinforcement Learning","date":"2019-05-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Sonkyunghwan/QTRAN","path":"Others/make_env.py","file_url":"https://github.com/Sonkyunghwan/QTRAN/blob/HEAD/Others/make_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ec3f64501ca07cda","mcp_get_code":{"code_sha256":"ec3f64501ca07cda"}},{"arxiv_id":"1810.02912","paper":"/paper/actor-attention-critic-for-multi-agent","title":"Actor-Attention-Critic for Multi-Agent Reinforcement Learning","date":"2018-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shariqiqbal2810/MAAC","path":"utils/make_env.py","file_url":"https://github.com/shariqiqbal2810/MAAC/blob/HEAD/utils/make_env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0fb08f714e49a890","mcp_get_code":{"code_sha256":"0fb08f714e49a890"}},{"arxiv_id":"1806.06923","paper":"/paper/implicit-quantile-networks-for-distributional","title":"Implicit Quantile Networks for Distributional Reinforcement Learning","date":"2018-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BY571/IQN","path":"wrapper.py","file_url":"https://github.com/BY571/IQN/blob/HEAD/wrapper.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"06fc02646fefa30a","mcp_get_code":{"code_sha256":"06fc02646fefa30a"}},{"arxiv_id":"1803.10122","paper":"/paper/world-models","title":"World Models","date":"2018-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hsgrandhi/AIProject","path":"env.py","file_url":"https://github.com/hsgrandhi/AIProject/blob/HEAD/env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ba7be13bc51628c9","mcp_get_code":{"code_sha256":"ba7be13bc51628c9"}},{"arxiv_id":"1803.00933","paper":"/paper/distributed-prioritized-experience-replay","title":"Distributed Prioritized Experience Replay","date":"2018-03-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haje01/distper","path":"wrappers.py","file_url":"https://github.com/haje01/distper/blob/HEAD/wrappers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"94cca620996f3fbd","mcp_get_code":{"code_sha256":"94cca620996f3fbd"}},{"arxiv_id":"1802.05438","paper":"/paper/mean-field-multi-agent-reinforcement-learning","title":"Mean Field Multi-Agent Reinforcement Learning","date":"2018-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baoqianwang/iros22_darl1n","path":"maddpg_o/experiments/train_normal.py","file_url":"https://github.com/baoqianwang/iros22_darl1n/blob/HEAD/maddpg_o/experiments/train_normal.py","status":"ran_fixture","verification_level":1,"contract_check":"DEP_MISSING","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e6e29c2c0e881c3c","mcp_get_code":{"code_sha256":"e6e29c2c0e881c3c"}},{"arxiv_id":"1802.05438","paper":"/paper/mean-field-multi-agent-reinforcement-learning","title":"Mean Field Multi-Agent Reinforcement Learning","date":"2018-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baoqianwang/iros22_darl1n","path":"maddpg_o/experiments/train_darl1n.py","file_url":"https://github.com/baoqianwang/iros22_darl1n/blob/HEAD/maddpg_o/experiments/train_darl1n.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b6a7ee165ede9526","mcp_get_code":{"code_sha256":"b6a7ee165ede9526"}},{"arxiv_id":"1703.03400","paper":"/paper/model-agnostic-meta-learning-for-fast","title":"Model-Agnostic Meta-Learning for Fast Adaptation of Deep Networks","date":"2017-03-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MoritzTaylor/maml-rl-tf2","path":"maml_rl/sampler.py","file_url":"https://github.com/MoritzTaylor/maml-rl-tf2/blob/HEAD/maml_rl/sampler.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4246e3602f23a27a","mcp_get_code":{"code_sha256":"4246e3602f23a27a"}},{"arxiv_id":"1609.05140","paper":"/paper/the-option-critic-architecture","title":"The Option-Critic Architecture","date":"2016-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AshrithSagar/option-critic","path":"oca/envs/utils.py","file_url":"https://github.com/AshrithSagar/option-critic/blob/HEAD/oca/envs/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e9c61c409e6c07d6","mcp_get_code":{"code_sha256":"e9c61c409e6c07d6"}},{"arxiv_id":"ijcai2022_0528","paper":null,"title":"arXiv:ijcai2022_0528","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"GyChou/mcppoElegantRLforCarla","path":"ray_elegantrl/interaction.py","file_url":"https://github.com/GyChou/mcppoElegantRLforCarla/blob/HEAD/ray_elegantrl/interaction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b917f33e17c81491","mcp_get_code":{"code_sha256":"b917f33e17c81491"}},{"arxiv_id":"aaai_29188","paper":null,"title":"arXiv:aaai_29188","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Jackory/RPBT","path":"ppo/ppo_data_collectors.py","file_url":"https://github.com/Jackory/RPBT/blob/HEAD/ppo/ppo_data_collectors.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7d3b32023c111d73","mcp_get_code":{"code_sha256":"7d3b32023c111d73"}},{"arxiv_id":"aaai_29188","paper":null,"title":"arXiv:aaai_29188","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Jackory/RPBT","path":"toyexample/rppo.py","file_url":"https://github.com/Jackory/RPBT/blob/HEAD/toyexample/rppo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c76cb158df6b31b2","mcp_get_code":{"code_sha256":"c76cb158df6b31b2"}}]}