{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/continuous-control/papers/5","list_of":"/task/continuous-control","task":"Continuous Control","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":12,"rows_per_page":100,"rows":[401,500],"of":1161,"counts":{"archive_papers_tagged":1161,"with_a_code_link":494,"where_syntology_ran_a_sample":193,"not_listed_spam_title":0,"listed":1161,"listed_where_code_ran":193,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/continuous-control","prev":"/task/continuous-control/papers/4","next":"/task/continuous-control/papers/6","papers":[{"url":"/paper/information-theoretic-regret-bounds-for","slug":"information-theoretic-regret-bounds-for","title":"Information Theoretic Regret Bounds for Online Nonlinear Control","date":"2020-06-22","arxiv_id":"2006.12466","repositories_listed":1,"syntology":null},{"url":"/paper/generating-adjacency-constrained-subgoals-in","slug":"generating-adjacency-constrained-subgoals-in","title":"Generating Adjacency-Constrained Subgoals in Hierarchical Reinforcement Learning","date":"2020-06-20","arxiv_id":"2006.11485","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generating-adjacency-constrained-subgoals-in#ran","syntology_url":"https://syntology.ai/paper/2006.11485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11485"}},"official":{"repos":["trzhang0116/HRAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/model-based-adversarial-meta-reinforcement","slug":"model-based-adversarial-meta-reinforcement","title":"Model-based Adversarial Meta-Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.08875","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-based-value-functions","slug":"parameter-based-value-functions","title":"Parameter-Based Value Functions","date":"2020-06-16","arxiv_id":"2006.09226","repositories_listed":1,"syntology":null},{"url":"/paper/qd-rl-efficient-mixing-of-quality-and","slug":"qd-rl-efficient-mixing-of-quality-and","title":"Diversity Policy Gradient for Sample Efficient Quality-Diversity Optimization","date":"2020-06-15","arxiv_id":"2006.08505","repositories_listed":1,"syntology":null},{"url":"/paper/a-policy-gradient-method-for-task-agnostic-1","slug":"a-policy-gradient-method-for-task-agnostic-1","title":"A Policy Gradient Method for Task-Agnostic Exploration","date":"2020-06-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dual-policy-distillation","slug":"dual-policy-distillation","title":"Dual Policy Distillation","date":"2020-06-07","arxiv_id":"2006.04061","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/dual-policy-distillation#ran","syntology_url":"https://syntology.ai/paper/2006.04061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04061"}},"official":null}},{"url":"/paper/prediction-with-directed-transitions-complex","slug":"prediction-with-directed-transitions-complex","title":"Prediction and Generalisation over Directed Actions by Grid Cells","date":"2020-06-05","arxiv_id":"2006.03355","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prediction-with-directed-transitions-complex#ran","syntology_url":"https://syntology.ai/paper/2006.03355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03355"}},"official":null}},{"url":"/paper/wasserstein-distance-guided-adversarial","slug":"wasserstein-distance-guided-adversarial","title":"Wasserstein Distance guided Adversarial Imitation Learning with Reward Shape Exploration","date":"2020-06-05","arxiv_id":"2006.03503","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-deep-reinforcement-learning-for","slug":"decentralized-deep-reinforcement-learning-for","title":"Decentralized Deep Reinforcement Learning for a Distributed and Adaptive Locomotion Controller of a Hexapod Robot","date":"2020-05-21","arxiv_id":"2005.11164","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-descent-policy-optimization","slug":"mirror-descent-policy-optimization","title":"Mirror Descent Policy Optimization","date":"2020-05-20","arxiv_id":"2005.09814","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-model-based-reinforcement","slug":"delay-aware-model-based-reinforcement","title":"Delay-Aware Model-Based Reinforcement Learning for Continuous Control","date":"2020-05-11","arxiv_id":"2005.05440","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-adversarial-inverse-reinforcement","slug":"off-policy-adversarial-inverse-reinforcement","title":"Off-Policy Adversarial Inverse Reinforcement Learning","date":"2020-05-03","arxiv_id":"2005.01138","repositories_listed":1,"syntology":null},{"url":"/paper/option-discovery-using-deep-skill-chaining","slug":"option-discovery-using-deep-skill-chaining","title":"Option Discovery using Deep Skill Chaining","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-guide-random-search-1","slug":"learning-to-guide-random-search-1","title":"Learning to Guide Random Search","date":"2020-04-25","arxiv_id":"2004.12214","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-to-guide-random-search-1#ran","syntology_url":"https://syntology.ai/paper/2004.12214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.12214"}},"official":{"repos":["intel-isl/LMRS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-reinforcement-learning-with-multi","slug":"continual-reinforcement-learning-with-multi","title":"Continual Reinforcement Learning with Multi-Timescale Replay","date":"2020-04-16","arxiv_id":"2004.07530","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/continual-reinforcement-learning-with-multi#ran","syntology_url":"https://syntology.ai/paper/2004.07530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07530"}},"official":{"repos":["ChristosKap/multi_timescale_replay"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-sparse-rewarded-tasks-from-sub","slug":"learning-sparse-rewarded-tasks-from-sub","title":"Learning Sparse Rewarded Tasks from Sub-Optimal Demonstrations","date":"2020-04-01","arxiv_id":"2004.00530","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-rewarded-tasks-from-sub#ran","syntology_url":"https://syntology.ai/paper/2004.00530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00530"}},"official":null}},{"url":"/paper/exploration-in-action-space","slug":"exploration-in-action-space","title":"Exploration in Action Space","date":"2020-03-31","arxiv_id":"2004.00500","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-investigation-of-the-challenges","slug":"an-empirical-investigation-of-the-challenges","title":"An empirical investigation of the challenges of real-world reinforcement learning","date":"2020-03-24","arxiv_id":"2003.11881","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-empirical-investigation-of-the-challenges#ran","syntology_url":"https://syntology.ai/paper/2003.11881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11881"}},"official":{"repos":["google-research/realworldrl_suite"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","repositories_listed":1,"syntology":null},{"url":"/paper/online-meta-critic-learning-for-off-policy-1","slug":"online-meta-critic-learning-for-off-policy-1","title":"Online Meta-Critic Learning for Off-Policy Actor-Critic Methods","date":"2020-03-11","arxiv_id":"2003.05334","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/online-meta-critic-learning-for-off-policy-1#ran","syntology_url":"https://syntology.ai/paper/2003.05334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05334"}},"official":{"repos":["zwfightzw/Meta-Critic"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-deep-reinforcement-learning-with","slug":"off-policy-deep-reinforcement-learning-with","title":"Off-Policy Deep Reinforcement Learning with Analogous Disentangled Exploration","date":"2020-02-25","arxiv_id":"2002.10738","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-for-probabilistic","slug":"safe-reinforcement-learning-for-probabilistic","title":"Safe reinforcement learning for probabilistic reachability and safety specifications: A Lyapunov-based approach","date":"2020-02-24","arxiv_id":"2002.10126","repositories_listed":1,"syntology":null},{"url":"/paper/variational-imitation-learning-with-diverse","slug":"variational-imitation-learning-with-diverse","title":"Variational Imitation Learning with Diverse-quality Demonstrations","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficacy-of-modern-neuro-evolutionary","slug":"efficacy-of-modern-neuro-evolutionary","title":"Efficacy of Modern Neuro-Evolutionary Strategies for Continuous Control Optimization","date":"2019-12-11","arxiv_id":"1912.05239","repositories_listed":1,"syntology":null},{"url":"/paper/better-exploration-with-optimistic-actor-1","slug":"better-exploration-with-optimistic-actor-1","title":"Better Exploration with Optimistic Actor Critic","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/loaded-dice-trading-off-bias-and-variance-in-1","slug":"loaded-dice-trading-off-bias-and-variance-in-1","title":"Loaded DiCE: Trading off Bias and Variance in Any-Order Score Function Gradient Estimators for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/smile-scalable-meta-inverse-reinforcement","slug":"smile-scalable-meta-inverse-reinforcement","title":"SMILe: Scalable Meta Inverse Reinforcement Learning through Context-Conditional Policies","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/behavior-regularized-offline-reinforcement-1","slug":"behavior-regularized-offline-reinforcement-1","title":"Behavior Regularized Offline Reinforcement Learning","date":"2019-11-26","arxiv_id":"1911.11361","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-task-agnostic-exploration-for","slug":"evaluating-task-agnostic-exploration-for","title":"Evaluating task-agnostic exploration for fixed-batch learning of arbitrary future tasks","date":"2019-11-20","arxiv_id":"1911.08666","repositories_listed":1,"syntology":null},{"url":"/paper/deep-tile-coder-an-efficient-sparse","slug":"deep-tile-coder-an-efficient-sparse","title":"Fuzzy Tiling Activations: A Simple Approach to Learning Sparse Representations Online","date":"2019-11-19","arxiv_id":"1911.08068","repositories_listed":1,"syntology":null},{"url":"/paper/online-replanning-in-belief-space-for","slug":"online-replanning-in-belief-space-for","title":"Online Replanning in Belief Space for Partially Observable Task and Motion Problems","date":"2019-11-11","arxiv_id":"1911.04577","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-skill-networks-unsupervised-robot","slug":"adversarial-skill-networks-unsupervised-robot","title":"Adversarial Skill Networks: Unsupervised Robot Skill Learning from Video","date":"2019-10-21","arxiv_id":"1910.09430","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-map-natural-language-instructions","slug":"learning-to-map-natural-language-instructions","title":"Learning to Map Natural Language Instructions to Physical Quadcopter Control using Simulated Flight","date":"2019-10-21","arxiv_id":"1910.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-map-natural-language-instructions#ran","syntology_url":"https://syntology.ai/paper/1910.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09664"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scoring-aggregating-planning-learning-task","slug":"scoring-aggregating-planning-learning-task","title":"Zero-shot Policy Learning with Spatial Temporal RewardDecomposition on Contingency-aware Observation","date":"2019-10-17","arxiv_id":"1910.08143","repositories_listed":1,"syntology":null},{"url":"/paper/policy-optimization-through-approximated","slug":"policy-optimization-through-approximated","title":"Policy Optimization Through Approximate Importance Sampling","date":"2019-10-09","arxiv_id":"1910.03857","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-optimization-through-approximated#ran","syntology_url":"https://syntology.ai/paper/1910.03857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.03857"}},"official":{"repos":["marctom/POTAIS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-sequential-monte-carlo-tunneling","slug":"dual-sequential-monte-carlo-tunneling","title":"DualSMC: Tunneling Differentiable Filtering and Planning under Continuous POMDPs","date":"2019-09-28","arxiv_id":"1909.13003","repositories_listed":1,"syntology":null},{"url":"/paper/the-differentiable-cross-entropy-method-1","slug":"the-differentiable-cross-entropy-method-1","title":"The Differentiable Cross-Entropy Method","date":"2019-09-27","arxiv_id":"1909.12830","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-differentiable-cross-entropy-method-1#ran","syntology_url":"https://syntology.ai/paper/1909.12830","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12830"}},"official":{"repos":["facebookresearch/dcem"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy","slug":"v-mpo-on-policy-maximum-a-posteriori-policy","title":"V-MPO: On-Policy Maximum a Posteriori Policy Optimization for Discrete and Continuous Control","date":"2019-09-26","arxiv_id":"1909.12238","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy#ran","syntology_url":"https://syntology.ai/paper/1909.12238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12238"}},"official":null}},{"url":"/paper/robel-robotics-benchmarks-for-learning-with","slug":"robel-robotics-benchmarks-for-learning-with","title":"ROBEL: Robotics Benchmarks for Learning with Low-Cost Robots","date":"2019-09-25","arxiv_id":"1909.11639","repositories_listed":1,"syntology":null},{"url":"/paper/loaded-dice-trading-off-bias-and-variance-in","slug":"loaded-dice-trading-off-bias-and-variance-in","title":"Loaded DiCE: Trading off Bias and Variance in Any-Order Score Function Estimators for Reinforcement Learning","date":"2019-09-23","arxiv_id":"1909.10549","repositories_listed":1,"syntology":null},{"url":"/paper/meta-inverse-reinforcement-learning-with","slug":"meta-inverse-reinforcement-learning-with","title":"Meta-Inverse Reinforcement Learning with Probabilistic Context Variables","date":"2019-09-20","arxiv_id":"1909.09314","repositories_listed":1,"syntology":null},{"url":"/paper/learning-action-transferable-policy-with","slug":"learning-action-transferable-policy-with","title":"Learning Action-Transferable Policy with Action Embedding","date":"2019-09-05","arxiv_id":"1909.02291","repositories_listed":1,"syntology":null},{"url":"/paper/learning-stabilizable-nonlinear-dynamics-with","slug":"learning-stabilizable-nonlinear-dynamics-with","title":"Learning Stabilizable Nonlinear Dynamics with Contraction-Based Regularization","date":"2019-07-29","arxiv_id":"1907.13122","repositories_listed":1,"syntology":null},{"url":"/paper/co-training-for-policy-learning","slug":"co-training-for-policy-learning","title":"Co-training for Policy Learning","date":"2019-07-03","arxiv_id":"1907.04484","repositories_listed":1,"syntology":null},{"url":"/paper/learning-belief-representations-for-imitation","slug":"learning-belief-representations-for-imitation","title":"Learning Belief Representations for Imitation Learning in POMDPs","date":"2019-06-22","arxiv_id":"1906.09510","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-learning-of-object-structure-and","slug":"unsupervised-learning-of-object-structure-and","title":"Unsupervised Learning of Object Structure and Dynamics from Videos","date":"2019-06-19","arxiv_id":"1906.07889","repositories_listed":1,"syntology":null},{"url":"/paper/mcp-learning-composable-hierarchical-control","slug":"mcp-learning-composable-hierarchical-control","title":"MCP: Learning Composable Hierarchical Control with Multiplicative Compositional Policies","date":"2019-05-23","arxiv_id":"1905.09808","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-exploration-in-off-policy","slug":"leveraging-exploration-in-off-policy","title":"Leveraging exploration in off-policy algorithms via normalizing flows","date":"2019-05-16","arxiv_id":"1905.06893","repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-as-task-inference","slug":"meta-reinforcement-learning-as-task-inference","title":"Meta reinforcement learning as task inference","date":"2019-05-15","arxiv_id":"1905.06424","repositories_listed":1,"syntology":null},{"url":"/paper/control-regularization-for-reduced-variance","slug":"control-regularization-for-reduced-variance","title":"Control Regularization for Reduced Variance Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05380","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/control-regularization-for-reduced-variance#ran","syntology_url":"https://syntology.ai/paper/1905.05380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05380"}},"official":{"repos":["rcheng805/CORE-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1905.00976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00976"}},"official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-time-mean-variance-portfolio","slug":"continuous-time-mean-variance-portfolio","title":"Continuous-Time Mean-Variance Portfolio Selection: A Reinforcement Learning Framework","date":"2019-04-25","arxiv_id":"1904.11392","repositories_listed":1,"syntology":null},{"url":"/paper/autoregressive-policies-for-continuous","slug":"autoregressive-policies-for-continuous","title":"Autoregressive Policies for Continuous Control Deep Reinforcement Learning","date":"2019-03-27","arxiv_id":"1903.11524","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-safe-reinforcement-learning","slug":"end-to-end-safe-reinforcement-learning","title":"End-to-End Safe Reinforcement Learning through Barrier Functions for Safety-Critical Continuous Control Tasks","date":"2019-03-21","arxiv_id":"1903.08792","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-episodic-deep-deterministic","slug":"asynchronous-episodic-deep-deterministic","title":"Asynchronous Episodic Deep Deterministic Policy Gradient: Towards Continuous Control in Computationally Complex Environments","date":"2019-03-03","arxiv_id":"1903.00827","repositories_listed":1,"syntology":null},{"url":"/paper/catalystrl-a-distributed-framework-for","slug":"catalystrl-a-distributed-framework-for","title":"Catalyst.RL: A Distributed Framework for Reproducible RL Research","date":"2019-02-28","arxiv_id":"1903.00027","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-bottlenecks-in-deep-q-learning","slug":"diagnosing-bottlenecks-in-deep-q-learning","title":"Diagnosing Bottlenecks in Deep Q-learning Algorithms","date":"2019-02-26","arxiv_id":"1902.10250","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diagnosing-bottlenecks-in-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1902.10250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.10250"}},"official":null}},{"url":"/paper/marathon-environments-multi-agent-continuous","slug":"marathon-environments-multi-agent-continuous","title":"Marathon Environments: Multi-Agent Continuous Control Benchmarks in a Modern Video Game Engine","date":"2019-02-25","arxiv_id":"1902.09097","repositories_listed":1,"syntology":null},{"url":"/paper/policy-consolidation-for-continual","slug":"policy-consolidation-for-continual","title":"Policy Consolidation for Continual Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00255","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-consolidation-for-continual#ran","syntology_url":"https://syntology.ai/paper/1902.00255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00255"}},"official":null}},{"url":"/paper/tf-replicator-distributed-machine-learning","slug":"tf-replicator-distributed-machine-learning","title":"TF-Replicator: Distributed Machine Learning for Researchers","date":"2019-02-01","arxiv_id":"1902.00465","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tf-replicator-distributed-machine-learning#ran","syntology_url":"https://syntology.ai/paper/1902.00465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00465"}},"official":{"repos":["tensorflow/community"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrasting-exploration-in-parameter-and","slug":"contrasting-exploration-in-parameter-and","title":"Contrasting Exploration in Parameter and Action Space: A Zeroth-Order Optimization Perspective","date":"2019-01-31","arxiv_id":"1901.11503","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-hierarchy-via-reinforcement","slug":"emergence-of-hierarchy-via-reinforcement","title":"Self-organization of action hierarchy and compositionality by reinforcement learning with recurrent neural networks","date":"2019-01-29","arxiv_id":"1901.10113","repositories_listed":1,"syntology":null},{"url":"/paper/lyapunov-based-safe-policy-optimization-for","slug":"lyapunov-based-safe-policy-optimization-for","title":"Lyapunov-based Safe Policy Optimization for Continuous Control","date":"2019-01-28","arxiv_id":"1901.10031","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-via","slug":"hierarchical-reinforcement-learning-via","title":"Hierarchical Reinforcement Learning via Advantage-Weighted Information Maximization","date":"2019-01-05","arxiv_id":"1901.01365","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-of-image-embedding","slug":"self-supervised-learning-of-image-embedding","title":"Self-supervised Learning of Image Embedding for Continuous Control","date":"2019-01-03","arxiv_id":"1901.00943","repositories_listed":1,"syntology":null},{"url":"/paper/relative-entropy-regularized-policy-iteration","slug":"relative-entropy-regularized-policy-iteration","title":"Relative Entropy Regularized Policy Iteration","date":"2018-12-05","arxiv_id":"1812.02256","repositories_listed":1,"syntology":null},{"url":"/paper/simple-random-search-of-static-linear","slug":"simple-random-search-of-static-linear","title":"Simple random search of static linear policies is competitive for reinforcement learning","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/model-learning-for-look-ahead-exploration-in","slug":"model-learning-for-look-ahead-exploration-in","title":"Model Learning for Look-ahead Exploration in Continuous Control","date":"2018-11-20","arxiv_id":"1811.08086","repositories_listed":1,"syntology":null},{"url":"/paper/mapping-navigation-instructions-to-continuous","slug":"mapping-navigation-instructions-to-continuous","title":"Mapping Navigation Instructions to Continuous Control Actions with Position-Visitation Prediction","date":"2018-11-10","arxiv_id":"1811.04179","repositories_listed":1,"syntology":null},{"url":"/paper/ace-an-actor-ensemble-algorithm-for","slug":"ace-an-actor-ensemble-algorithm-for","title":"ACE: An Actor Ensemble Algorithm for Continuous Control with Tree Search","date":"2018-11-06","arxiv_id":"1811.02696","repositories_listed":1,"syntology":null},{"url":"/paper/inverse-reinforcement-learning-for-video","slug":"inverse-reinforcement-learning-for-video","title":"Inverse reinforcement learning for video games","date":"2018-10-24","arxiv_id":"1810.10593","repositories_listed":1,"syntology":null},{"url":"/paper/ppo-cma-proximal-policy-optimization-with","slug":"ppo-cma-proximal-policy-optimization-with","title":"PPO-CMA: Proximal Policy Optimization with Covariance Matrix Adaptation","date":"2018-10-05","arxiv_id":"1810.02541","repositories_listed":1,"syntology":null},{"url":"/paper/where-did-my-optimum-go-an-empirical-analysis","slug":"where-did-my-optimum-go-an-empirical-analysis","title":"Where Did My Optimum Go?: An Empirical Analysis of Gradient Descent Optimization in Policy Gradient Methods","date":"2018-10-05","arxiv_id":"1810.02525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/where-did-my-optimum-go-an-empirical-analysis#ran","syntology_url":"https://syntology.ai/paper/1810.02525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02525"}},"official":{"repos":["facebookresearch/WhereDidMyOptimumGo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emi-exploration-with-mutual-information","slug":"emi-exploration-with-mutual-information","title":"EMI: Exploration with Mutual Information","date":"2018-10-02","arxiv_id":"1810.01176","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emi-exploration-with-mutual-information#ran","syntology_url":"https://syntology.ai/paper/1810.01176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01176"}},"official":{"repos":["snu-mllab/EMI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/archer-aggressive-rewards-to-counter-bias-in","slug":"archer-aggressive-rewards-to-counter-bias-in","title":"ARCHER: Aggressive Rewards to Counter bias in Hindsight Experience Replay","date":"2018-09-06","arxiv_id":"1809.02070","repositories_listed":1,"syntology":null},{"url":"/paper/a-tour-of-reinforcement-learning-the-view","slug":"a-tour-of-reinforcement-learning-the-view","title":"A Tour of Reinforcement Learning: The View from Continuous Control","date":"2018-06-25","arxiv_id":"1806.09460","repositories_listed":1,"syntology":null},{"url":"/paper/multi-objective-model-based-policy-search-for","slug":"multi-objective-model-based-policy-search-for","title":"Multi-objective Model-based Policy Search for Data-efficient Learning with Sparse Rewards","date":"2018-06-25","arxiv_id":"1806.09351","repositories_listed":1,"syntology":null},{"url":"/paper/barc-backward-reachability-curriculum-for","slug":"barc-backward-reachability-curriculum-for","title":"BaRC: Backward Reachability Curriculum for Robotic Reinforcement Learning","date":"2018-06-16","arxiv_id":"1806.06161","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/barc-backward-reachability-curriculum-for#ran","syntology_url":"https://syntology.ai/paper/1806.06161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.06161"}},"official":{"repos":["StanfordASL/BaRC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/marginal-policy-gradients-a-unified-family-of","slug":"marginal-policy-gradients-a-unified-family-of","title":"Marginal Policy Gradients: A Unified Family of Estimators for Bounded Action Spaces with Applications","date":"2018-06-13","arxiv_id":"1806.05134","repositories_listed":1,"syntology":null},{"url":"/paper/policy-optimization-with-second-order","slug":"policy-optimization-with-second-order","title":"Policy Optimization with Second-Order Advantage Information","date":"2018-05-09","arxiv_id":"1805.03586","repositories_listed":1,"syntology":null},{"url":"/paper/clipped-action-policy-gradient","slug":"clipped-action-policy-gradient","title":"Clipped Action Policy Gradient","date":"2018-02-21","arxiv_id":"1802.07564","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clipped-action-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/1802.07564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.07564"}},"official":{"repos":["pfnet-research/capg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nervenet-learning-structured-policy-with","slug":"nervenet-learning-structured-policy-with","title":"NerveNet: Learning Structured Policy with Graph Neural Networks","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-policy-gradients-via-alpha","slug":"bayesian-policy-gradients-via-alpha","title":"Bayesian Policy Gradients via Alpha Divergence Dropout Inference","date":"2017-12-06","arxiv_id":"1712.02037","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-ddpg-method-with-prioritized","slug":"a-novel-ddpg-method-with-prioritized","title":"A novel DDPG method with prioritized experience replay","date":"2017-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/optiongan-learning-joint-reward-policy","slug":"optiongan-learning-joint-reward-policy","title":"OptionGAN: Learning Joint Reward-Policy Options using Generative Adversarial Inverse Reinforcement Learning","date":"2017-09-20","arxiv_id":"1709.06683","repositories_listed":1,"syntology":null},{"url":"/paper/using-parameterized-black-box-priors-to-scale","slug":"using-parameterized-black-box-priors-to-scale","title":"Using Parameterized Black-Box Priors to Scale Up Model-Based Policy Search for Robotics","date":"2017-09-20","arxiv_id":"1709.06917","repositories_listed":1,"syntology":null},{"url":"/paper/reproducibility-of-benchmarked-deep","slug":"reproducibility-of-benchmarked-deep","title":"Reproducibility of Benchmarked Deep Reinforcement Learning Tasks for Continuous Control","date":"2017-08-10","arxiv_id":"1708.04133","repositories_listed":1,"syntology":null},{"url":"/paper/rail-risk-averse-imitation-learning","slug":"rail-risk-averse-imitation-learning","title":"RAIL: Risk-Averse Imitation Learning","date":"2017-07-20","arxiv_id":"1707.06658","repositories_listed":1,"syntology":null},{"url":"/paper/trust-pcl-an-off-policy-trust-region-method","slug":"trust-pcl-an-off-policy-trust-region-method","title":"Trust-PCL: An Off-Policy Trust Region Method for Continuous Control","date":"2017-07-06","arxiv_id":"1707.01891","repositories_listed":1,"syntology":null},{"url":"/paper/guide-actor-critic-for-continuous-control","slug":"guide-actor-critic-for-continuous-control","title":"Guide Actor-Critic for Continuous Control","date":"2017-05-22","arxiv_id":"1705.07606","repositories_listed":1,"syntology":null},{"url":"/paper/black-box-data-efficient-policy-search-for","slug":"black-box-data-efficient-policy-search-for","title":"Black-Box Data-efficient Policy Search for Robotics","date":"2017-03-21","arxiv_id":"1703.07261","repositories_listed":1,"syntology":null},{"url":"/paper/towards-generalization-and-simplicity-in","slug":"towards-generalization-and-simplicity-in","title":"Towards Generalization and Simplicity in Continuous Control","date":"2017-03-08","arxiv_id":"1703.02660","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-pivoting-task","slug":"reinforcement-learning-for-pivoting-task","title":"Reinforcement Learning for Pivoting Task","date":"2017-03-01","arxiv_id":"1703.00472","repositories_listed":1,"syntology":null},{"url":null,"slug":"supervised-fine-tuning-on-curated-data-is","title":"Supervised Fine Tuning on Curated Data is Reinforcement Learning (and can be improved)","date":"2025-07-17","arxiv_id":"2507.12856","repositories_listed":0,"syntology":null},{"url":null,"slug":"rqdia-regularizing-q-value-distributions-with-1","title":"rQdia: Regularizing Q-Value Distributions With Image Augmentation","date":"2025-06-26","arxiv_id":"2506.21367","repositories_listed":0,"syntology":null},{"url":null,"slug":"fractional-reasoning-via-latent-steering","title":"Fractional Reasoning via Latent Steering Vectors Improves Inference Time Compute","date":"2025-06-18","arxiv_id":"2506.15882","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-algorithm-distillation-for-continuous","title":"Scaling Algorithm Distillation for Continuous Control with Mamba","date":"2025-06-16","arxiv_id":"2506.13892","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10167","title":"Wasserstein Barycenter Soft Actor-Critic","date":"2025-06-11","arxiv_id":"2506.10167","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-implicit-imitation","title":"Reinforcement Learning via Implicit Imitation Guidance","date":"2025-06-09","arxiv_id":"2506.07505","repositories_listed":0,"syntology":null}],"record_sha256":"f6ec1e8d9a70d0e333fdb02667adaf057e2c2101a14f6301975dd8b75532d3d1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}