{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/43","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":43,"pages_in_order":152,"rows_per_page":100,"rows":[4201,4300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/42","next":"/task/reinforcement-learning-1/papers/44","papers":[{"url":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proximal-distilled-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09807"}},"official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ranking-policy-gradient","slug":"ranking-policy-gradient","title":"Ranking Policy Gradient","date":"2019-06-24","arxiv_id":"1906.09674","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ranking-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/1906.09674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09674"}},"official":{"repos":["illidanlab/rpg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-neurally-plausible-model-learns-successor","slug":"a-neurally-plausible-model-learns-successor","title":"A neurally plausible model learns successor representations in partially observable environments","date":"2019-06-22","arxiv_id":"1906.09480","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/a-neurally-plausible-model-learns-successor#ran","syntology_url":"https://syntology.ai/paper/1906.09480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09480"}},"official":{"repos":["evertes/distributional_SF"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reinforcement-learning-models-of-human","slug":"reinforcement-learning-models-of-human","title":"A Story of Two Streams: Reinforcement Learning Models from Human Behavior and Neuropsychiatry","date":"2019-06-21","arxiv_id":"1906.11286","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-convex","slug":"reinforcement-learning-with-convex","title":"Reinforcement Learning with Convex Constraints","date":"2019-06-21","arxiv_id":"1906.09323","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-convex#ran","syntology_url":"https://syntology.ai/paper/1906.09323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09323"}},"official":{"repos":["xkianteb/ApproPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/split-q-learning-reinforcement-learning-with","slug":"split-q-learning-reinforcement-learning-with","title":"Split Q Learning: Reinforcement Learning with Two-Stream Rewards","date":"2019-06-21","arxiv_id":"1906.12350","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-for-1","slug":"a-deep-reinforcement-learning-approach-for-1","title":"A Deep Reinforcement Learning Approach for Global Routing","date":"2019-06-20","arxiv_id":"1906.08809","repositories_listed":1,"syntology":null},{"url":"/paper/a-hierarchical-architecture-for-sequential","slug":"a-hierarchical-architecture-for-sequential","title":"A Hierarchical Architecture for Sequential Decision-Making in Autonomous Driving using Deep Reinforcement Learning","date":"2019-06-20","arxiv_id":"1906.08464","repositories_listed":1,"syntology":null},{"url":"/paper/calibrated-model-based-deep-reinforcement","slug":"calibrated-model-based-deep-reinforcement","title":"Calibrated Model-Based Deep Reinforcement Learning","date":"2019-06-19","arxiv_id":"1906.08312","repositories_listed":1,"syntology":null},{"url":"/paper/learning-driven-exploration-for-reinforcement","slug":"learning-driven-exploration-for-reinforcement","title":"Learning-Driven Exploration for Reinforcement Learning","date":"2019-06-17","arxiv_id":"1906.06890","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-industrial","slug":"deep-reinforcement-learning-for-industrial","title":"Deep Reinforcement Learning for Industrial Insertion Tasks with Visual Inputs and Natural Rewards","date":"2019-06-13","arxiv_id":"1906.05841","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-for-industrial#ran","syntology_url":"https://syntology.ai/paper/1906.05841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05841"}},"official":null}},{"url":"/paper/goal-conditioned-imitation-learning","slug":"goal-conditioned-imitation-learning","title":"Goal-conditioned Imitation Learning","date":"2019-06-13","arxiv_id":"1906.05838","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goal-conditioned-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1906.05838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05838"}},"official":{"repos":["dingyiming0427/goalgail"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-knowledge-graph-reasoning-for","slug":"reinforcement-knowledge-graph-reasoning-for","title":"Reinforcement Knowledge Graph Reasoning for Explainable Recommendation","date":"2019-06-12","arxiv_id":"1906.05237","repositories_listed":1,"syntology":null},{"url":"/paper/search-on-the-replay-buffer-bridging-planning","slug":"search-on-the-replay-buffer-bridging-planning","title":"Search on the Replay Buffer: Bridging Planning and Reinforcement Learning","date":"2019-06-12","arxiv_id":"1906.05253","repositories_listed":1,"syntology":null},{"url":"/paper/causal-discovery-with-reinforcement-learning","slug":"causal-discovery-with-reinforcement-learning","title":"Causal Discovery with Reinforcement Learning","date":"2019-06-11","arxiv_id":"1906.04477","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-channel-coding","slug":"reinforcement-learning-for-channel-coding","title":"Reinforcement Learning for Channel Coding: Learned Bit-Flipping Decoding","date":"2019-06-11","arxiv_id":"1906.04448","repositories_listed":1,"syntology":null},{"url":"/paper/wasserstein-reinforcement-learning","slug":"wasserstein-reinforcement-learning","title":"Learning to Score Behaviors for Guided Policy Optimization","date":"2019-06-11","arxiv_id":"1906.04349","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-via-hindsight-goal-generation","slug":"exploration-via-hindsight-goal-generation","title":"Exploration via Hindsight Goal Generation","date":"2019-06-10","arxiv_id":"1906.04279","repositories_listed":1,"syntology":null},{"url":"/paper/neural-keyphrase-generation-via-reinforcement","slug":"neural-keyphrase-generation-via-reinforcement","title":"Neural Keyphrase Generation via Reinforcement Learning with Adaptive Rewards","date":"2019-06-10","arxiv_id":"1906.04106","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-keyphrase-generation-via-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04106"}},"official":{"repos":["kenchan0226/keyphrase-generation-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/curiosity-driven-multi-criteria-hindsight","slug":"curiosity-driven-multi-criteria-hindsight","title":"Curiosity-Driven Multi-Criteria Hindsight Experience Replay","date":"2019-06-09","arxiv_id":"1906.03710","repositories_listed":1,"syntology":null},{"url":"/paper/gossip-based-actor-learner-architectures-for","slug":"gossip-based-actor-learner-architectures-for","title":"Gossip-based Actor-Learner Architectures for Deep Reinforcement Learning","date":"2019-06-09","arxiv_id":"1906.04585","repositories_listed":1,"syntology":null},{"url":"/paper/svrg-for-policy-evaluation-with-fewer","slug":"svrg-for-policy-evaluation-with-fewer","title":"SVRG for Policy Evaluation with Fewer Gradient Evaluations","date":"2019-06-09","arxiv_id":"1906.03704","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/svrg-for-policy-evaluation-with-fewer#ran","syntology_url":"https://syntology.ai/paper/1906.03704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.03704"}},"official":null}},{"url":"/paper/ego-pose-estimation-and-forecasting-as-real","slug":"ego-pose-estimation-and-forecasting-as-real","title":"Ego-Pose Estimation and Forecasting as Real-Time PD Control","date":"2019-06-07","arxiv_id":"1906.03173","repositories_listed":1,"syntology":null},{"url":"/paper/preference-based-interactive-multi-document","slug":"preference-based-interactive-multi-document","title":"Preference-based Interactive Multi-Document Summarisation","date":"2019-06-07","arxiv_id":"1906.02923","repositories_listed":1,"syntology":null},{"url":"/paper/improving-exploration-in-soft-actor-critic","slug":"improving-exploration-in-soft-actor-critic","title":"Improving Exploration in Soft-Actor-Critic with Normalizing Flows Policies","date":"2019-06-06","arxiv_id":"1906.02771","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-exploration-in-soft-actor-critic#ran","syntology_url":"https://syntology.ai/paper/1906.02771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.02771"}},"official":{"repos":["joeybose/FloRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-interpretable-reinforcement-learning","slug":"towards-interpretable-reinforcement-learning","title":"Towards Interpretable Reinforcement Learning Using Attention Augmented Agents","date":"2019-06-06","arxiv_id":"1906.02500","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-when-all-actions-are","slug":"reinforcement-learning-when-all-actions-are","title":"Reinforcement Learning When All Actions are Not Always Available","date":"2019-06-05","arxiv_id":"1906.01772","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-with-low-complexity#ran","syntology_url":"https://syntology.ai/paper/1906.01695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01695"}},"official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-exploration-in-linear-quadratic","slug":"robust-exploration-in-linear-quadratic","title":"Robust exploration in linear quadratic reinforcement learning","date":"2019-06-04","arxiv_id":"1906.01584","repositories_listed":1,"syntology":null},{"url":"/paper/190600584","slug":"190600584","title":"A Semi-Supervised Approach for Low-Resourced Text Generation","date":"2019-06-03","arxiv_id":"1906.00584","repositories_listed":1,"syntology":null},{"url":"/paper/190600889","slug":"190600889","title":"Learning to solve the credit assignment problem","date":"2019-06-03","arxiv_id":"1906.00889","repositories_listed":1,"syntology":null},{"url":"/paper/190600421","slug":"190600421","title":"Air Learning: A Deep Reinforcement Learning Gym for Autonomous Aerial Robot Visual Navigation","date":"2019-06-02","arxiv_id":"1906.00421","repositories_listed":1,"syntology":null},{"url":"/paper/190600422","slug":"190600422","title":"On the Correctness and Sample Complexity of Inverse Reinforcement Learning","date":"2019-06-02","arxiv_id":"1906.00422","repositories_listed":1,"syntology":null},{"url":"/paper/190600214","slug":"190600214","title":"Harnessing Reinforcement Learning for Neural Motion Planning","date":"2019-06-01","arxiv_id":"1906.00214","repositories_listed":1,"syntology":null},{"url":"/paper/extending-deep-model-predictive-control-with","slug":"extending-deep-model-predictive-control-with","title":"Safety Augmented Value Estimation from Demonstrations (SAVED): Safe Deep Model-Based RL for Sparse Cost Robotic Tasks","date":"2019-05-31","arxiv_id":"1905.13402","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extending-deep-model-predictive-control-with#ran","syntology_url":"https://syntology.ai/paper/1905.13402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13402"}},"official":null}},{"url":"/paper/interval-timing-in-deep-reinforcement","slug":"interval-timing-in-deep-reinforcement","title":"Interval timing in deep reinforcement learning agents","date":"2019-05-31","arxiv_id":"1905.13469","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-modeling-of-temporal-credit","slug":"sequence-modeling-of-temporal-credit","title":"Sequence Modeling of Temporal Credit Assignment for Episodic Reinforcement Learning","date":"2019-05-31","arxiv_id":"1905.13420","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequence-modeling-of-temporal-credit#ran","syntology_url":"https://syntology.ai/paper/1905.13420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13420"}},"official":null}},{"url":"/paper/reinforcement-learning-and-adaptive-sampling","slug":"reinforcement-learning-and-adaptive-sampling","title":"Reinforcement Learning and Adaptive Sampling for Optimized DNN Compilation","date":"2019-05-30","arxiv_id":"1905.12799","repositories_listed":1,"syntology":null},{"url":"/paper/towards-finding-longer-proofs","slug":"towards-finding-longer-proofs","title":"Towards Finding Longer Proofs","date":"2019-05-30","arxiv_id":"1905.13100","repositories_listed":1,"syntology":null},{"url":"/paper/190513547","slug":"190513547","title":"Learning robust control for LQR systems with multiplicative noise via policy gradient","date":"2019-05-28","arxiv_id":"1905.13547","repositories_listed":1,"syntology":null},{"url":"/paper/coordinated-exploration-via-intrinsic-rewards","slug":"coordinated-exploration-via-intrinsic-rewards","title":"Coordinated Exploration via Intrinsic Rewards for Multi-Agent Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.12127","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coordinated-exploration-via-intrinsic-rewards#ran","syntology_url":"https://syntology.ai/paper/1905.12127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.12127"}},"official":{"repos":["shariqiqbal2810/Multi-Explore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/snooping-attacks-on-deep-reinforcement","slug":"snooping-attacks-on-deep-reinforcement","title":"Snooping Attacks on Deep Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.11832","repositories_listed":1,"syntology":null},{"url":"/paper/finite-time-analysis-of-q-learning-with","slug":"finite-time-analysis-of-q-learning-with","title":"Finite-Sample Analysis of Nonlinear Stochastic Approximation with Applications in Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11425","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finite-time-analysis-of-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/1905.11425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11425"}},"official":null}},{"url":"/paper/learning-to-discretize-solving-1d-scalar","slug":"learning-to-discretize-solving-1d-scalar","title":"Learning to Discretize: Solving 1D Scalar Conservation Laws via Deep Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11079","repositories_listed":1,"syntology":null},{"url":"/paper/tight-regret-bounds-for-model-based","slug":"tight-regret-bounds-for-model-based","title":"Tight Regret Bounds for Model-Based Reinforcement Learning with Greedy Policies","date":"2019-05-27","arxiv_id":"1905.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tight-regret-bounds-for-model-based#ran","syntology_url":"https://syntology.ai/paper/1905.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11527"}},"official":{"repos":["NMerlis/TabulaRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-kernel-loss-for-solving-the-bellman","slug":"a-kernel-loss-for-solving-the-bellman","title":"A Kernel Loss for Solving the Bellman Equation","date":"2019-05-25","arxiv_id":"1905.10506","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-via-online","slug":"safe-reinforcement-learning-via-online","title":"Safe Reinforcement Learning with Nonlinear Dynamics via Model Predictive Shielding","date":"2019-05-25","arxiv_id":"1905.10691","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-symmetric-reward-noising-for","slug":"adaptive-symmetric-reward-noising-for","title":"Adaptive Symmetric Reward Noising for Reinforcement Learning","date":"2019-05-24","arxiv_id":"1905.10144","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-via-flow-based-intrinsic-rewards","slug":"exploration-via-flow-based-intrinsic-rewards","title":"Exploration via Flow-Based Intrinsic Rewards","date":"2019-05-24","arxiv_id":"1905.10071","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-for-1","slug":"hierarchical-reinforcement-learning-for-1","title":"Hierarchical Reinforcement Learning for Concurrent Discovery of Compound and Composable Policies","date":"2019-05-23","arxiv_id":"1905.09668","repositories_listed":1,"syntology":null},{"url":"/paper/multi-hop-reading-comprehension-via-deep","slug":"multi-hop-reading-comprehension-via-deep","title":"Multi-hop Reading Comprehension via Deep Reinforcement Learning based Document Traversal","date":"2019-05-23","arxiv_id":"1905.09438","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-parameter","slug":"deep-reinforcement-learning-based-parameter","title":"Deep Reinforcement Learning Based Parameter Control in Differential Evolution","date":"2019-05-20","arxiv_id":"1905.08006","repositories_listed":1,"syntology":null},{"url":"/paper/a-regularized-opponent-model-with-maximum","slug":"a-regularized-opponent-model-with-maximum","title":"A Regularized Opponent Model with Maximum Entropy Objective","date":"2019-05-17","arxiv_id":"1905.08087","repositories_listed":1,"syntology":null},{"url":"/paper/exact-k-recommendation-via-maximal-clique","slug":"exact-k-recommendation-via-maximal-clique","title":"Exact-K Recommendation via Maximal Clique Optimization","date":"2019-05-17","arxiv_id":"1905.07089","repositories_listed":1,"syntology":null},{"url":"/paper/mastering-the-game-of-sungka-from-random-play","slug":"mastering-the-game-of-sungka-from-random-play","title":"Mastering the Game of Sungka from Random Play","date":"2019-05-17","arxiv_id":"1905.07102","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-exploration-in-off-policy","slug":"leveraging-exploration-in-off-policy","title":"Leveraging exploration in off-policy algorithms via normalizing flows","date":"2019-05-16","arxiv_id":"1905.06893","repositories_listed":1,"syntology":null},{"url":"/paper/qbso-fs-a-reinforcement-learning-based-bee","slug":"qbso-fs-a-reinforcement-learning-based-bee","title":"QBSO-FS: A Reinforcement Learning Based Bee Swarm Optimization Metaheuristic for Feature Selection","date":"2019-05-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/expressive-priors-in-bayesian-neural-networks","slug":"expressive-priors-in-bayesian-neural-networks","title":"Expressive Priors in Bayesian Neural Networks: Kernel Combinations and Periodic Functions","date":"2019-05-15","arxiv_id":"1905.06076","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/expressive-priors-in-bayesian-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1905.06076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.06076"}},"official":null}},{"url":"/paper/meta-reinforcement-learning-as-task-inference","slug":"meta-reinforcement-learning-as-task-inference","title":"Meta reinforcement learning as task inference","date":"2019-05-15","arxiv_id":"1905.06424","repositories_listed":1,"syntology":null},{"url":"/paper/control-regularization-for-reduced-variance","slug":"control-regularization-for-reduced-variance","title":"Control Regularization for Reduced Variance Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05380","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/control-regularization-for-reduced-variance#ran","syntology_url":"https://syntology.ai/paper/1905.05380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05380"}},"official":{"repos":["rcheng805/CORE-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/successor-options-an-option-discovery","slug":"successor-options-an-option-discovery","title":"Successor Options: An Option Discovery Framework for Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05731","repositories_listed":1,"syntology":null},{"url":"/paper/cityflow-a-multi-agent-reinforcement-learning","slug":"cityflow-a-multi-agent-reinforcement-learning","title":"CityFlow: A Multi-Agent Reinforcement Learning Environment for Large Scale City Traffic Scenario","date":"2019-05-13","arxiv_id":"1905.05217","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cityflow-a-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1905.05217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05217"}},"official":{"repos":["cityflow-project/CityFlow"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-image-classification-via","slug":"multi-agent-image-classification-via","title":"Multi-Agent Image Classification via Reinforcement Learning","date":"2019-05-13","arxiv_id":"1905.04835","repositories_listed":1,"syntology":null},{"url":"/paper/task-agnostic-dynamics-priors-for-deep","slug":"task-agnostic-dynamics-priors-for-deep","title":"Task-Agnostic Dynamics Priors for Deep Reinforcement Learning","date":"2019-05-13","arxiv_id":"1905.04819","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-agnostic-dynamics-priors-for-deep#ran","syntology_url":"https://syntology.ai/paper/1905.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04819"}},"official":{"repos":["yilundu/task_agnostic_dynamics_prior"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/learning-phase-competition-for-traffic-signal","slug":"learning-phase-competition-for-traffic-signal","title":"Learning Phase Competition for Traffic Signal Control","date":"2019-05-12","arxiv_id":"1905.04722","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-phase-competition-for-traffic-signal#ran","syntology_url":"https://syntology.ai/paper/1905.04722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04722"}},"official":null}},{"url":"/paper/autonomous-management-of-energy-harvesting","slug":"autonomous-management-of-energy-harvesting","title":"Autonomous Management of Energy-Harvesting IoT Nodes Using Deep Reinforcement Learning","date":"2019-05-10","arxiv_id":"1905.04181","repositories_listed":1,"syntology":null},{"url":"/paper/190503389","slug":"190503389","title":"Learning to Evolve","date":"2019-05-08","arxiv_id":"1905.03389","repositories_listed":1,"syntology":null},{"url":"/paper/dimension-wise-importance-sampling-weight","slug":"dimension-wise-importance-sampling-weight","title":"Dimension-Wise Importance Sampling Weight Clipping for Sample-Efficient Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.02363","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":12,"n_pointer_only":18,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dimension-wise-importance-sampling-weight#ran","syntology_url":"https://syntology.ai/paper/1905.02363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.02363"}},"official":{"repos":["seungyulhan/disc"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-ordinal-reinforcement-learning","slug":"deep-ordinal-reinforcement-learning","title":"Deep Ordinal Reinforcement Learning","date":"2019-05-06","arxiv_id":"1905.02005","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-control-in-metric-space-with","slug":"learning-to-control-in-metric-space-with","title":"Learning to Control in Metric Space with Optimal Regret","date":"2019-05-05","arxiv_id":"1905.01576","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-in-metric-space-with#ran","syntology_url":"https://syntology.ai/paper/1905.01576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.01576"}},"official":null}},{"url":"/paper/p3o-policy-on-policy-off-policy-optimization","slug":"p3o-policy-on-policy-off-policy-optimization","title":"P3O: Policy-on Policy-off Policy Optimization","date":"2019-05-05","arxiv_id":"1905.01756","repositories_listed":1,"syntology":null},{"url":"/paper/the-game-of-tetris-in-machine-learning","slug":"the-game-of-tetris-in-machine-learning","title":"The Game of Tetris in Machine Learning","date":"2019-05-05","arxiv_id":"1905.01652","repositories_listed":1,"syntology":null},{"url":"/paper/deep-residual-reinforcement-learning","slug":"deep-residual-reinforcement-learning","title":"Deep Residual Reinforcement Learning","date":"2019-05-03","arxiv_id":"1905.01072","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1905.00976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00976"}},"official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dher-hindsight-experience-replay-for-dynamic","slug":"dher-hindsight-experience-replay-for-dynamic","title":"DHER: Hindsight Experience Replay for Dynamic Goals","date":"2019-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficient-model-free-reinforcement-learning-1","slug":"efficient-model-free-reinforcement-learning-1","title":"Efficient Model-free Reinforcement Learning in Metric Spaces","date":"2019-05-01","arxiv_id":"1905.00475","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-policy-update","slug":"supervised-policy-update","title":"SUPERVISED POLICY UPDATE","date":"2019-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/challenges-of-real-world-reinforcement","slug":"challenges-of-real-world-reinforcement","title":"Challenges of Real-World Reinforcement Learning","date":"2019-04-29","arxiv_id":"1904.12901","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-time-mean-variance-portfolio","slug":"continuous-time-mean-variance-portfolio","title":"Continuous-Time Mean-Variance Portfolio Selection: A Reinforcement Learning Framework","date":"2019-04-25","arxiv_id":"1904.11392","repositories_listed":1,"syntology":null},{"url":"/paper/neural-logic-reinforcement-learning","slug":"neural-logic-reinforcement-learning","title":"Neural Logic Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.10729","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neural-logic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1904.10729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10729"}},"official":{"repos":["ZhengyaoJiang/NLRL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/graphnas-graph-neural-architecture-search","slug":"graphnas-graph-neural-architecture-search","title":"GraphNAS: Graph Neural Architecture Search with Reinforcement Learning","date":"2019-04-22","arxiv_id":"1904.09981","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/graphnas-graph-neural-architecture-search#ran","syntology_url":"https://syntology.ai/paper/1904.09981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09981"}},"official":{"repos":["GraphNAS/GraphNAS-simple"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-minerl-competition-on-sample-efficient","slug":"the-minerl-competition-on-sample-efficient","title":"The MineRL 2019 Competition on Sample Efficient Reinforcement Learning using Human Priors","date":"2019-04-22","arxiv_id":"1904.10079","repositories_listed":1,"syntology":null},{"url":"/paper/190501360","slug":"190501360","title":"Skynet: A Top Deep RL Agent in the Inaugural Pommerman Team Competition","date":"2019-04-20","arxiv_id":"1905.01360","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-compositional-language-with-deep","slug":"emergence-of-compositional-language-with-deep","title":"Emergence of Compositional Language with Deep Generational Transmission","date":"2019-04-19","arxiv_id":"1904.09067","repositories_listed":1,"syntology":null},{"url":"/paper/posterior-regularized-reinforce-for-instance","slug":"posterior-regularized-reinforce-for-instance","title":"Posterior-regularized REINFORCE for Instance Selection in Distant Supervision","date":"2019-04-17","arxiv_id":"1904.08051","repositories_listed":1,"syntology":null},{"url":"/paper/towards-robust-deep-reinforcement-learning","slug":"towards-robust-deep-reinforcement-learning","title":"Towards Robust Deep Reinforcement Learning for Traffic Signal Control: Demand Surges, Incidents and Sensor Failures","date":"2019-04-17","arxiv_id":"1904.08353","repositories_listed":1,"syntology":null},{"url":"/paper/lets-play-again-variability-of-deep","slug":"lets-play-again-variability-of-deep","title":"Let's Play Again: Variability of Deep Reinforcement Learning Agents in Atari Environments","date":"2019-04-12","arxiv_id":"1904.06312","repositories_listed":1,"syntology":null},{"url":"/paper/reinbo-machine-learning-pipeline-search-and","slug":"reinbo-machine-learning-pipeline-search-and","title":"ReinBo: Machine Learning pipeline search and configuration with Bayesian Optimization embedded Reinforcement Learning","date":"2019-04-10","arxiv_id":"1904.05381","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-on-a-budget-3d","slug":"deep-reinforcement-learning-on-a-budget-3d","title":"Deep Reinforcement Learning on a Budget: 3D Control and Reasoning Without a Supercomputer","date":"2019-04-03","arxiv_id":"1904.01806","repositories_listed":1,"syntology":null},{"url":"/paper/jointly-pre-training-with-supervised","slug":"jointly-pre-training-with-supervised","title":"Jointly Pre-training with Supervised, Autoencoder, and Value Losses for Deep Reinforcement Learning","date":"2019-04-03","arxiv_id":"1904.02206","repositories_listed":1,"syntology":null},{"url":"/paper/random-projection-in-neural-episodic-control","slug":"random-projection-in-neural-episodic-control","title":"Random Projection in Neural Episodic Control","date":"2019-04-03","arxiv_id":"1904.01790","repositories_listed":1,"syntology":null},{"url":"/paper/dynamically-optimal-treatment-allocation","slug":"dynamically-optimal-treatment-allocation","title":"Dynamically Optimal Treatment Allocation","date":"2019-04-01","arxiv_id":"1904.01047","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-pick-the-domain-randomization","slug":"how-to-pick-the-domain-randomization","title":"How to pick the domain randomization parameters for sim-to-real transfer of reinforcement learning policies?","date":"2019-03-28","arxiv_id":"1903.11774","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/how-to-pick-the-domain-randomization#ran","syntology_url":"https://syntology.ai/paper/1903.11774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.11774"}},"official":{"repos":["quanvuong/domain_randomization"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/autoregressive-policies-for-continuous","slug":"autoregressive-policies-for-continuous","title":"Autoregressive Policies for Continuous Control Deep Reinforcement Learning","date":"2019-03-27","arxiv_id":"1903.11524","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-off-policy-actor-critic","slug":"generalized-off-policy-actor-critic","title":"Generalized Off-Policy Actor-Critic","date":"2019-03-27","arxiv_id":"1903.11329","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-regression-for-constructing-analytic","slug":"symbolic-regression-for-constructing-analytic","title":"Constructing Parsimonious Analytic Models for Dynamic Systems via Symbolic Regression","date":"2019-03-27","arxiv_id":"1903.11483","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-safe-reinforcement-learning","slug":"end-to-end-safe-reinforcement-learning","title":"End-to-End Safe Reinforcement Learning through Barrier Functions for Safety-Critical Continuous Control Tasks","date":"2019-03-21","arxiv_id":"1903.08792","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-thermodynamic-trajectories-using","slug":"optimizing-thermodynamic-trajectories-using","title":"Optimizing thermodynamic trajectories using evolutionary and gradient-based reinforcement learning","date":"2019-03-20","arxiv_id":"1903.08543","repositories_listed":1,"syntology":null},{"url":"/paper/a-multi-agent-off-policy-actor-critic","slug":"a-multi-agent-off-policy-actor-critic","title":"A Multi-Agent Off-Policy Actor-Critic Algorithm for Distributed Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06372","repositories_listed":1,"syntology":null},{"url":"/paper/gym-gazebo2-a-toolkit-for-reinforcement","slug":"gym-gazebo2-a-toolkit-for-reinforcement","title":"gym-gazebo2, a toolkit for reinforcement learning using ROS 2 and Gazebo","date":"2019-03-14","arxiv_id":"1903.06278","repositories_listed":1,"syntology":null}],"record_sha256":"c41210d7f372076c03df6a7a6da33273f2aea9857035922e7084a82f883c4877","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}