{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mujoco/papers/3","list_of":"/task/mujoco","task":"MuJoCo","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":7,"rows_per_page":100,"rows":[201,300],"of":677,"counts":{"archive_papers_tagged":677,"with_a_code_link":293,"where_syntology_ran_a_sample":115,"not_listed_spam_title":0,"listed":677,"listed_where_code_ran":115,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":103,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":103,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mujoco","prev":"/task/mujoco/papers/2","next":"/task/mujoco/papers/4","papers":[{"url":"/paper/a-pragmatic-look-at-deep-imitation-learning","slug":"a-pragmatic-look-at-deep-imitation-learning","title":"A Pragmatic Look at Deep Imitation Learning","date":"2021-08-04","arxiv_id":"2108.01867","repositories_listed":1,"syntology":null},{"url":"/paper/tianshou-a-highly-modularized-deep","slug":"tianshou-a-highly-modularized-deep","title":"Tianshou: a Highly Modularized Deep Reinforcement Learning Library","date":"2021-07-29","arxiv_id":"2107.14171","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tianshou-a-highly-modularized-deep#ran","syntology_url":"https://syntology.ai/paper/2107.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14171"}},"official":{"repos":["thu-ml/tianshou"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-offline-distributional","slug":"conservative-offline-distributional","title":"Conservative Offline Distributional Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.06106","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-offline-distributional#ran","syntology_url":"https://syntology.ai/paper/2107.06106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06106"}},"official":{"repos":["JasonMa2016/CODAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-mutual-information-mummi-training","slug":"multi-modal-mutual-information-mummi-training","title":"Multi-Modal Mutual Information (MuMMI) Training for Robust Self-Supervised Deep Reinforcement Learning","date":"2021-07-06","arxiv_id":"2107.02339","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-adversarial-attacks-on-1","slug":"understanding-adversarial-attacks-on-1","title":"Understanding Adversarial Attacks on Observations in Deep Reinforcement Learning","date":"2021-06-30","arxiv_id":"2106.15860","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-skill-discovery-with-bottleneck","slug":"unsupervised-skill-discovery-with-bottleneck","title":"Unsupervised Skill Discovery with Bottleneck Option Learning","date":"2021-06-27","arxiv_id":"2106.14305","repositories_listed":1,"syntology":null},{"url":"/paper/brax-a-differentiable-physics-engine-for","slug":"brax-a-differentiable-physics-engine-for","title":"Brax -- A Differentiable Physics Engine for Large Scale Rigid Body Simulation","date":"2021-06-24","arxiv_id":"2106.13281","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/brax-a-differentiable-physics-engine-for#ran","syntology_url":"https://syntology.ai/paper/2106.13281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13281"}},"official":{"repos":["google/brax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-safe-reinforcement-learning-via-1","slug":"towards-safe-reinforcement-learning-via-1","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value at Risk","date":"2021-06-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-to-4","slug":"a-deep-reinforcement-learning-approach-to-4","title":"A Deep Reinforcement Learning Approach to Marginalized Importance Sampling with the Successor Representation","date":"2021-06-12","arxiv_id":"2106.06854","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-deep-reinforcement-learning-approach-to-4#ran","syntology_url":"https://syntology.ai/paper/2106.06854","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06854"}},"official":{"repos":["sfujim/SR-DICE"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-game-theoretic-approach-to-multi-agent","slug":"a-game-theoretic-approach-to-multi-agent","title":"A Game-Theoretic Approach to Multi-Agent Trust Region Optimization","date":"2021-06-12","arxiv_id":"2106.06828","repositories_listed":1,"syntology":null},{"url":"/paper/who-is-the-strongest-enemy-towards-optimal","slug":"who-is-the-strongest-enemy-towards-optimal","title":"Who Is the Strongest Enemy? Towards Optimal and Efficient Evasion Attacks in Deep RL","date":"2021-06-09","arxiv_id":"2106.05087","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":1,"n_instrument":8,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/who-is-the-strongest-enemy-towards-optimal#ran","syntology_url":"https://syntology.ai/paper/2106.05087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05087"}},"official":{"repos":["umd-huang-lab/paad_adv_rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-covariate-shift-in-imitation","slug":"mitigating-covariate-shift-in-imitation","title":"Mitigating Covariate Shift in Imitation Learning via Offline Data Without Great Coverage","date":"2021-06-06","arxiv_id":"2106.03207","repositories_listed":1,"syntology":null},{"url":"/paper/improving-generalization-in-meta-rl-with","slug":"improving-generalization-in-meta-rl-with","title":"Improving Generalization in Meta-RL with Imaginary Tasks from Latent Dynamics Mixture","date":"2021-05-28","arxiv_id":"2105.13524","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/improving-generalization-in-meta-rl-with#ran","syntology_url":"https://syntology.ai/paper/2105.13524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13524"}},"official":{"repos":["suyoung-lee/ldm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/mitigating-covariate-shift-in-imitation-1","slug":"mitigating-covariate-shift-in-imitation-1","title":"Mitigating Covariate Shift in Imitation Learning via Offline Data With Partial Coverage","date":"2021-05-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-experience-replay","slug":"regret-minimization-experience-replay","title":"Regret Minimization Experience Replay in Off-Policy Reinforcement Learning","date":"2021-05-15","arxiv_id":"2105.07253","repositories_listed":1,"syntology":null},{"url":"/paper/context-based-soft-actor-critic-for","slug":"context-based-soft-actor-critic-for","title":"Context-Based Soft Actor Critic for Environments with Non-stationary Dynamics","date":"2021-05-07","arxiv_id":"2105.03310","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-mixture-of-experts-for-1","slug":"probabilistic-mixture-of-experts-for-1","title":"Probabilistic Mixture-of-Experts for Efficient Deep Reinforcement Learning","date":"2021-04-19","arxiv_id":"2104.09122","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probabilistic-mixture-of-experts-for-1#ran","syntology_url":"https://syntology.ai/paper/2104.09122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09122"}},"official":{"repos":["JieRen98/rlkit-pmoe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-what-to-do-by-simulating-the-past-1","slug":"learning-what-to-do-by-simulating-the-past-1","title":"Learning What To Do by Simulating the Past","date":"2021-04-08","arxiv_id":"2104.03946","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-what-to-do-by-simulating-the-past-1#ran","syntology_url":"https://syntology.ai/paper/2104.03946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.03946"}},"official":{"repos":["HumanCompatibleAI/deep-rlsp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/no-need-for-interactions-robust-model-based","slug":"no-need-for-interactions-robust-model-based","title":"No Need for Interactions: Robust Model-Based Imitation Learning using Neural ODE","date":"2021-04-03","arxiv_id":"2104.01390","repositories_listed":1,"syntology":null},{"url":"/paper/a-quadratic-actor-network-for-model-free","slug":"a-quadratic-actor-network-for-model-free","title":"A Quadratic Actor Network for Model-Free Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06617","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-episodic-memory-for-deep","slug":"generalizable-episodic-memory-for-deep","title":"Generalizable Episodic Memory for Deep Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06469","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-episodic-memory-for-deep#ran","syntology_url":"https://syntology.ai/paper/2103.06469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.06469"}},"official":{"repos":["MouseHu/GEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-free-policy-learning-with-reward","slug":"model-free-policy-learning-with-reward","title":"Model-free Policy Learning with Reward Gradients","date":"2021-03-09","arxiv_id":"2103.05147","repositories_listed":1,"syntology":null},{"url":"/paper/q-value-weighted-regression-reinforcement-1","slug":"q-value-weighted-regression-reinforcement-1","title":"Q-Value Weighted Regression: Reinforcement Learning with Limited Data","date":"2021-02-12","arxiv_id":"2102.06782","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/q-value-weighted-regression-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2102.06782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06782"}},"official":null}},{"url":"/paper/robust-policy-gradient-against-strong-data","slug":"robust-policy-gradient-against-strong-data","title":"Robust Policy Gradient against Strong Data Corruption","date":"2021-02-11","arxiv_id":"2102.05800","repositories_listed":1,"syntology":null},{"url":"/paper/variance-penalized-on-policy-and-off-policy","slug":"variance-penalized-on-policy-and-off-policy","title":"Variance Penalized On-Policy and Off-Policy Actor-Critic","date":"2021-02-03","arxiv_id":"2102.01985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/variance-penalized-on-policy-and-off-policy#ran","syntology_url":"https://syntology.ai/paper/2102.01985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.01985"}},"official":{"repos":["arushi12130/VariancePenalizedActorCritic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-domain-adaptation-for","slug":"cross-modal-domain-adaptation-for","title":"Cross-Modal Domain Adaptation for Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-trust-region-learning","slug":"multi-agent-trust-region-learning","title":"Multi-Agent Trust Region Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/teac-intergrating-trust-region-and-max","slug":"teac-intergrating-trust-region-and-max","title":"TEAC: Intergrating Trust Region and Max Entropy Actor Critic for Continuous Control","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/locally-persistent-exploration-in-continuous","slug":"locally-persistent-exploration-in-continuous","title":"Locally Persistent Exploration in Continuous Control Tasks with Sparse Rewards","date":"2020-12-26","arxiv_id":"2012.13658","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/locally-persistent-exploration-in-continuous#ran","syntology_url":"https://syntology.ai/paper/2012.13658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.13658"}},"official":{"repos":["h-aboutalebi/SparseBaseline"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reset-free-lifelong-learning-with-skill-space-1","slug":"reset-free-lifelong-learning-with-skill-space-1","title":"Reset-Free Lifelong Learning with Skill-Space Planning","date":"2020-12-07","arxiv_id":"2012.03548","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-transition-improving-sample","slug":"continuous-transition-improving-sample","title":"Continuous Transition: Improving Sample Efficiency for Continuous Control Problems via MixUp","date":"2020-11-30","arxiv_id":"2011.14487","repositories_listed":1,"syntology":null},{"url":"/paper/realant-an-open-source-low-cost-quadruped-for","slug":"realant-an-open-source-low-cost-quadruped-for","title":"RealAnt: An Open-Source Low-Cost Quadruped for Education and Research in Real-World Reinforcement Learning","date":"2020-11-05","arxiv_id":"2011.03085","repositories_listed":1,"syntology":null},{"url":"/paper/human-guided-robot-behavior-learning-a-gan","slug":"human-guided-robot-behavior-learning-a-gan","title":"Human-guided Robot Behavior Learning: A GAN-assisted Preference-based Reinforcement Learning Approach","date":"2020-10-15","arxiv_id":"2010.07467","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-deep-reinforcement-learning-with-1","slug":"multi-task-deep-reinforcement-learning-with-1","title":"Knowledge Transfer in Multi-Task Deep Reinforcement Learning for Continuous Control","date":"2020-10-15","arxiv_id":"2010.07494","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2010.07494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07494"}},"official":null}},{"url":"/paper/self-imitation-learning-in-sparse-reward","slug":"self-imitation-learning-in-sparse-reward","title":"Self-Imitation Learning for Robot Tasks with Sparse and Delayed Rewards","date":"2020-10-14","arxiv_id":"2010.06962","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-design-choices-in-proximal-policy","slug":"revisiting-design-choices-in-proximal-policy","title":"Revisiting Design Choices in Proximal Policy Optimization","date":"2020-09-23","arxiv_id":"2009.10897","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/revisiting-design-choices-in-proximal-policy#ran","syntology_url":"https://syntology.ai/paper/2009.10897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.10897"}},"official":{"repos":["chloechsu/revisiting-ppo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-automated-deep-reinforcement","slug":"sample-efficient-automated-deep-reinforcement","title":"Sample-Efficient Automated Deep Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01555","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-automated-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.01555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.01555"}},"official":{"repos":["automl/SEARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-with-sinkhorn-distances","slug":"imitation-learning-with-sinkhorn-distances","title":"Imitation Learning with Sinkhorn Distances","date":"2020-08-20","arxiv_id":"2008.09167","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-variational-model-based","slug":"contrastive-variational-model-based","title":"Contrastive Variational Reinforcement Learning for Complex Observations","date":"2020-08-06","arxiv_id":"2008.02430","repositories_listed":1,"syntology":null},{"url":"/paper/human-preference-scaling-with-demonstrations","slug":"human-preference-scaling-with-demonstrations","title":"Weak Human Preference Supervision For Deep Reinforcement Learning","date":"2020-07-25","arxiv_id":"2007.12904","repositories_listed":1,"syntology":null},{"url":"/paper/nengo-and-low-power-ai-hardware-for-robust","slug":"nengo-and-low-power-ai-hardware-for-robust","title":"Nengo and low-power AI hardware for robust, embedded neurorobotics","date":"2020-07-20","arxiv_id":"2007.10227","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-cup-and-ball-with-noisy","slug":"learning-to-play-cup-and-ball-with-noisy","title":"Learning to Play Cup-and-Ball with Noisy Camera Observations","date":"2020-07-19","arxiv_id":"2007.09562","repositories_listed":1,"syntology":null},{"url":"/paper/fast-adaptation-via-policy-dynamics-value","slug":"fast-adaptation-via-policy-dynamics-value","title":"Fast Adaptation via Policy-Dynamics Value Functions","date":"2020-07-06","arxiv_id":"2007.02879","repositories_listed":1,"syntology":null},{"url":"/paper/meta-sac-auto-tune-the-entropy-temperature-of","slug":"meta-sac-auto-tune-the-entropy-temperature-of","title":"Meta-SAC: Auto-tune the Entropy Temperature of Soft Actor-Critic via Metagradient","date":"2020-07-03","arxiv_id":"2007.01932","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/meta-sac-auto-tune-the-entropy-temperature-of#ran","syntology_url":"https://syntology.ai/paper/2007.01932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01932"}},"official":{"repos":["twni2016/Meta-SAC"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dm-control-software-and-tasks-for-continuous","slug":"dm-control-software-and-tasks-for-continuous","title":"dm_control: Software and Tasks for Continuous Control","date":"2020-06-22","arxiv_id":"2006.12983","repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-effectively-explore-in-context-based","slug":"learn-to-effectively-explore-in-context-based","title":"MetaCURE: Meta Reinforcement Learning with Empowerment-Driven Exploration","date":"2020-06-15","arxiv_id":"2006.08170","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learn-to-effectively-explore-in-context-based#ran","syntology_url":"https://syntology.ai/paper/2006.08170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08170"}},"official":{"repos":["NagisaZj/MetaCURE-Public"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wasserstein-distance-guided-adversarial","slug":"wasserstein-distance-guided-adversarial","title":"Wasserstein Distance guided Adversarial Imitation Learning with Reward Shape Exploration","date":"2020-06-05","arxiv_id":"2006.03503","repositories_listed":1,"syntology":null},{"url":"/paper/novel-policy-seeking-with-constrained","slug":"novel-policy-seeking-with-constrained","title":"Novel Policy Seeking with Constrained Optimization","date":"2020-05-21","arxiv_id":"2005.10696","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-model-based-reinforcement","slug":"delay-aware-model-based-reinforcement","title":"Delay-Aware Model-Based Reinforcement Learning for Continuous Control","date":"2020-05-11","arxiv_id":"2005.05440","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-stochastic-policy-distillation","slug":"evolutionary-stochastic-policy-distillation","title":"Evolutionary Stochastic Policy Distillation","date":"2020-04-27","arxiv_id":"2004.12909","repositories_listed":1,"syntology":null},{"url":"/paper/per-step-reward-a-new-perspective-for-risk","slug":"per-step-reward-a-new-perspective-for-risk","title":"Mean-Variance Policy Iteration for Risk-Averse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10888","repositories_listed":1,"syntology":null},{"url":"/paper/state-only-imitation-with-transition-dynamics-1","slug":"state-only-imitation-with-transition-dynamics-1","title":"State-only Imitation with Transition Dynamics Mismatch","date":"2020-02-27","arxiv_id":"2002.11879","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/state-only-imitation-with-transition-dynamics-1#ran","syntology_url":"https://syntology.ai/paper/2002.11879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.11879"}},"official":{"repos":["tgangwani/RL-Indirect-imitation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-via-adversarial-1","slug":"robust-reinforcement-learning-via-adversarial-1","title":"Robust Reinforcement Learning via Adversarial training with Langevin Dynamics","date":"2020-02-14","arxiv_id":"2002.06063","repositories_listed":1,"syntology":null},{"url":"/paper/periodic-intra-ensemble-knowledge","slug":"periodic-intra-ensemble-knowledge","title":"Periodic Intra-Ensemble Knowledge Distillation for Reinforcement Learning","date":"2020-02-01","arxiv_id":"2002.00149","repositories_listed":1,"syntology":null},{"url":"/paper/deep-tile-coder-an-efficient-sparse","slug":"deep-tile-coder-an-efficient-sparse","title":"Fuzzy Tiling Activations: A Simple Approach to Learning Sparse Representations Online","date":"2019-11-19","arxiv_id":"1911.08068","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-methods-for-model-based","slug":"asynchronous-methods-for-model-based","title":"Asynchronous Methods for Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12453","repositories_listed":1,"syntology":null},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-variational-inference-and-pac-bayes","slug":"unifying-variational-inference-and-pac-bayes","title":"Unifying Variational Inference and PAC-Bayes for Supervised Learning that Scales","date":"2019-10-23","arxiv_id":"1910.10367","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapping-the-expressivity-with-model","slug":"bootstrapping-the-expressivity-with-model","title":"On the Expressivity of Neural Networks for Deep Reinforcement Learning","date":"2019-10-14","arxiv_id":"1910.05927","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-with-3","slug":"hierarchical-reinforcement-learning-with-3","title":"Hierarchical Reinforcement Learning with Advantage-Based Auxiliary Rewards","date":"2019-10-10","arxiv_id":"1910.04450","repositories_listed":1,"syntology":null},{"url":"/paper/formal-language-constraints-for-markov","slug":"formal-language-constraints-for-markov","title":"Formal Language Constraints for Markov Decision Processes","date":"2019-10-02","arxiv_id":"1910.01074","repositories_listed":1,"syntology":null},{"url":"/paper/a-generalized-training-approach-for-1","slug":"a-generalized-training-approach-for-1","title":"A Generalized Training Approach for Multiagent Learning","date":"2019-09-27","arxiv_id":"1909.12823","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapping-the-expressivity-with-model-1","slug":"bootstrapping-the-expressivity-with-model-1","title":"Bootstrapping the Expressivity with Model-based Planning","date":"2019-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mdp-playground-meta-features-in-reinforcement","slug":"mdp-playground-meta-features-in-reinforcement","title":"MDP Playground: An Analysis and Debug Testbed for Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.07750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-playground-meta-features-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1909.07750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07750"}},"official":{"repos":["automl/mdp-playground"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regularized-anderson-acceleration-for-off","slug":"regularized-anderson-acceleration-for-off","title":"Regularized Anderson Acceleration for Off-Policy Deep Reinforcement Learning","date":"2019-09-07","arxiv_id":"1909.03245","repositories_listed":1,"syntology":null},{"url":"/paper/towards-model-based-reinforcement-learning","slug":"towards-model-based-reinforcement-learning","title":"Towards Model-based Reinforcement Learning for Industry-near Environments","date":"2019-07-27","arxiv_id":"1907.11971","repositories_listed":1,"syntology":null},{"url":"/paper/orrb-openai-remote-rendering-backend","slug":"orrb-openai-remote-rendering-backend","title":"ORRB -- OpenAI Remote Rendering Backend","date":"2019-06-26","arxiv_id":"1906.11633","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-model-based-planning-with-policy","slug":"exploring-model-based-planning-with-policy","title":"Exploring Model-based Planning with Policy Networks","date":"2019-06-20","arxiv_id":"1906.08649","repositories_listed":1,"syntology":null},{"url":"/paper/calibrated-model-based-deep-reinforcement","slug":"calibrated-model-based-deep-reinforcement","title":"Calibrated Model-Based Deep Reinforcement Learning","date":"2019-06-19","arxiv_id":"1906.08312","repositories_listed":1,"syntology":null},{"url":"/paper/learning-powerful-policies-by-using","slug":"learning-powerful-policies-by-using","title":"Learning Powerful Policies by Using Consistent Dynamics Model","date":"2019-06-11","arxiv_id":"1906.04355","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-modeling-of-temporal-credit","slug":"sequence-modeling-of-temporal-credit","title":"Sequence Modeling of Temporal Credit Assignment for Episodic Reinforcement Learning","date":"2019-05-31","arxiv_id":"1905.13420","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequence-modeling-of-temporal-credit#ran","syntology_url":"https://syntology.ai/paper/1905.13420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13420"}},"official":null}},{"url":"/paper/leveraging-exploration-in-off-policy","slug":"leveraging-exploration-in-off-policy","title":"Leveraging exploration in off-policy algorithms via normalizing flows","date":"2019-05-16","arxiv_id":"1905.06893","repositories_listed":1,"syntology":null},{"url":"/paper/p3o-policy-on-policy-off-policy-optimization","slug":"p3o-policy-on-policy-off-policy-optimization","title":"P3O: Policy-on Policy-off Policy Optimization","date":"2019-05-05","arxiv_id":"1905.01756","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1905.00976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00976"}},"official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/supervised-policy-update","slug":"supervised-policy-update","title":"SUPERVISED POLICY UPDATE","date":"2019-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/generalized-off-policy-actor-critic","slug":"generalized-off-policy-actor-critic","title":"Generalized Off-Policy Actor-Critic","date":"2019-03-27","arxiv_id":"1903.11329","repositories_listed":1,"syntology":null},{"url":"/paper/-rank-multi-agent-evaluation-by-evolution","slug":"-rank-multi-agent-evaluation-by-evolution","title":"α-Rank: Multi-Agent Evaluation by Evolution","date":"2019-03-04","arxiv_id":"1903.01373","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-episodic-deep-deterministic","slug":"asynchronous-episodic-deep-deterministic","title":"Asynchronous Episodic Deep Deterministic Policy Gradient: Towards Continuous Control in Computationally Complex Environments","date":"2019-03-03","arxiv_id":"1903.00827","repositories_listed":1,"syntology":null},{"url":"/paper/lyapunov-based-safe-policy-optimization-for","slug":"lyapunov-based-safe-policy-optimization-for","title":"Lyapunov-based Safe Policy Optimization for Continuous Control","date":"2019-01-28","arxiv_id":"1901.10031","repositories_listed":1,"syntology":null},{"url":"/paper/wall-e-an-efficient-reinforcement-learning","slug":"wall-e-an-efficient-reinforcement-learning","title":"WALL-E: An Efficient Reinforcement Learning Research Framework","date":"2019-01-18","arxiv_id":"1901.06086","repositories_listed":1,"syntology":null},{"url":"/paper/residual-policy-learning","slug":"residual-policy-learning","title":"Residual Policy Learning","date":"2018-12-15","arxiv_id":"1812.06298","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/residual-policy-learning#ran","syntology_url":"https://syntology.ai/paper/1812.06298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.06298"}},"official":null}},{"url":"/paper/simple-random-search-of-static-linear","slug":"simple-random-search-of-static-linear","title":"Simple random search of static linear policies is competitive for reinforcement learning","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/episodic-curiosity-through-reachability","slug":"episodic-curiosity-through-reachability","title":"Episodic Curiosity through Reachability","date":"2018-10-04","arxiv_id":"1810.02274","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/episodic-curiosity-through-reachability#ran","syntology_url":"https://syntology.ai/paper/1810.02274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02274"}},"official":{"repos":["google-research/episodic-curiosity"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/torille-learning-environment-for-hand-to-hand","slug":"torille-learning-environment-for-hand-to-hand","title":"ToriLLE: Learning Environment for Hand-to-Hand Combat","date":"2018-07-26","arxiv_id":"1807.10110","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-policy-update-for-deep","slug":"supervised-policy-update-for-deep","title":"Supervised Policy Update for Deep Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11706","repositories_listed":1,"syntology":null},{"url":"/paper/policy-optimization-with-second-order","slug":"policy-optimization-with-second-order","title":"Policy Optimization with Second-Order Advantage Information","date":"2018-05-09","arxiv_id":"1805.03586","repositories_listed":1,"syntology":null},{"url":"/paper/on-learning-intrinsic-rewards-for-policy","slug":"on-learning-intrinsic-rewards-for-policy","title":"On Learning Intrinsic Rewards for Policy Gradient Methods","date":"2018-04-17","arxiv_id":"1804.06459","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/on-learning-intrinsic-rewards-for-policy#ran","syntology_url":"https://syntology.ai/paper/1804.06459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.06459"}},"official":{"repos":["Hwhitetooth/lirpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/back-to-basics-benchmarking-canonical","slug":"back-to-basics-benchmarking-canonical","title":"Back to Basics: Benchmarking Canonical Evolution Strategies for Playing Atari","date":"2018-02-24","arxiv_id":"1802.08842","repositories_listed":1,"syntology":null},{"url":"/paper/structured-control-nets-for-deep","slug":"structured-control-nets-for-deep","title":"Structured Control Nets for Deep Reinforcement Learning","date":"2018-02-22","arxiv_id":"1802.08311","repositories_listed":1,"syntology":null},{"url":"/paper/nervenet-learning-structured-policy-with","slug":"nervenet-learning-structured-policy-with","title":"NerveNet: Learning Structured Policy with Graph Neural Networks","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-policy-gradients-via-alpha","slug":"bayesian-policy-gradients-via-alpha","title":"Bayesian Policy Gradients via Alpha Divergence Dropout Inference","date":"2017-12-06","arxiv_id":"1712.02037","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-ddpg-method-with-prioritized","slug":"a-novel-ddpg-method-with-prioritized","title":"A novel DDPG method with prioritized experience replay","date":"2017-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mujoco-a-physics-engine-for-model-based","slug":"mujoco-a-physics-engine-for-model-based","title":"MuJoCo: A physics engine for model-based control","date":"2012-10-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"aligning-humans-and-robots-via-reinforcement","title":"Aligning Humans and Robots via Reinforcement Learning from Implicit Human Feedback","date":"2025-07-17","arxiv_id":"2507.13171","repositories_listed":0,"syntology":null},{"url":null,"slug":"turning-sand-to-gold-recycling-data-to-bridge","title":"Turning Sand to Gold: Recycling Data to Bridge On-Policy and Off-Policy Learning via Causal Bound","date":"2025-07-15","arxiv_id":"2507.11269","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-and-mitigating-reward-hacking-in","title":"Detecting and Mitigating Reward Hacking in Reinforcement Learning Systems: A Comprehensive Empirical Study","date":"2025-07-08","arxiv_id":"2507.05619","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-domain-randomization-via-uncertainty","title":"Safe Domain Randomization via Uncertainty-Aware Out-of-Distribution Detection and Policy Adaptation","date":"2025-07-08","arxiv_id":"2507.06111","repositories_listed":0,"syntology":null},{"url":null,"slug":"rqdia-regularizing-q-value-distributions-with-1","title":"rQdia: Regularizing Q-Value Distributions With Image Augmentation","date":"2025-06-26","arxiv_id":"2506.21367","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-expert-performance-with-limited","title":"Beyond-Expert Performance with Limited Demonstrations: Efficient Imitation Learning with Double Exploration","date":"2025-06-25","arxiv_id":"2506.20307","repositories_listed":0,"syntology":null},{"url":null,"slug":"hard-contacts-with-soft-gradients-refining","title":"Hard Contacts with Soft Gradients: Refining Differentiable Simulators for Learning and Control","date":"2025-06-17","arxiv_id":"2506.14186","repositories_listed":0,"syntology":null}],"record_sha256":"287a3c322f0aa70f9b0e92d884b41c4dd336153c582056699201c32bd22950fa","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}