{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/41","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":41,"pages_in_order":152,"rows_per_page":100,"rows":[4001,4100],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/40","next":"/task/reinforcement-learning-1/papers/42","papers":[{"url":"/paper/increasing-performance-of-electric-vehicles","slug":"increasing-performance-of-electric-vehicles","title":"Increasing performance of electric vehicles in ride-hailing services using deep reinforcement learning","date":"2019-12-07","arxiv_id":"1912.03408","repositories_listed":1,"syntology":null},{"url":"/paper/valan-vision-and-language-agent-navigation","slug":"valan-vision-and-language-agent-navigation","title":"VALAN: Vision and Language Agent Navigation","date":"2019-12-06","arxiv_id":"1912.03241","repositories_listed":1,"syntology":null},{"url":"/paper/191202368","slug":"191202368","title":"Inter-Level Cooperation in Hierarchical Reinforcement Learning","date":"2019-12-05","arxiv_id":"1912.02368","repositories_listed":1,"syntology":null},{"url":"/paper/blind-inpainting-of-large-scale-masks-of-thin","slug":"blind-inpainting-of-large-scale-masks-of-thin","title":"Blind Inpainting of Large-scale Masks of Thin Structures with Adversarial and Reinforcement Learning","date":"2019-12-05","arxiv_id":"1912.02470","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-credit-assignment-1","slug":"hindsight-credit-assignment-1","title":"Hindsight Credit Assignment","date":"2019-12-05","arxiv_id":"1912.02503","repositories_listed":1,"syntology":null},{"url":"/paper/mo-states-mo-problems-emergency-stop-1","slug":"mo-states-mo-problems-emergency-stop-1","title":"Mo' States Mo' Problems: Emergency Stop Mechanisms from Observation","date":"2019-12-03","arxiv_id":"1912.01649","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-farsighted-agents-tend-to-seek-power","slug":"optimal-farsighted-agents-tend-to-seek-power","title":"Optimal Policies Tend to Seek Power","date":"2019-12-03","arxiv_id":"1912.01683","repositories_listed":1,"syntology":null},{"url":"/paper/safelife-10-exploring-side-effects-in-complex","slug":"safelife-10-exploring-side-effects-in-complex","title":"SafeLife 1.0: Exploring Side Effects in Complex Environments","date":"2019-12-03","arxiv_id":"1912.01217","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safelife-10-exploring-side-effects-in-complex#ran","syntology_url":"https://syntology.ai/paper/1912.01217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01217"}},"official":{"repos":["PartnershipOnAI/safelife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-model-based-reinforcement-learning-with-1","slug":"a-model-based-reinforcement-learning-with-1","title":"A Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-auxiliary-task-weighting-for","slug":"adaptive-auxiliary-task-weighting-for","title":"Adaptive Auxiliary Task Weighting for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-device-placement","slug":"learning-generalizable-device-placement","title":"Learning Generalizable Device Placement Algorithms for Distributed Machine Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-local-search-heuristics-for-boolean","slug":"learning-local-search-heuristics-for-boolean","title":"Learning Local Search Heuristics for Boolean Satisfiability","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-reward-machines-for-partially","slug":"learning-reward-machines-for-partially","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/liir-learning-individual-intrinsic-reward-in","slug":"liir-learning-individual-intrinsic-reward-in","title":"LIIR: Learning Individual Intrinsic Reward in Multi-Agent Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/loaded-dice-trading-off-bias-and-variance-in-1","slug":"loaded-dice-trading-off-bias-and-variance-in-1","title":"Loaded DiCE: Trading off Bias and Variance in Any-Order Score Function Gradient Estimators for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/non-stationary-markov-decision-processes-a-1","slug":"non-stationary-markov-decision-processes-a-1","title":"Non-Stationary Markov Decision Processes, a Worst-Case Approach using Model-Based Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/park-an-open-platform-for-learning-augmented","slug":"park-an-open-platform-for-learning-augmented","title":"Park: An Open Platform for Learning-Augmented Computer Systems","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/privacy-preserving-q-learning-with-functional","slug":"privacy-preserving-q-learning-with-functional","title":"Privacy-Preserving Q-Learning with Functional Noise in Continuous Spaces","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/propagating-uncertainty-in-reinforcement","slug":"propagating-uncertainty-in-reinforcement","title":"Propagating Uncertainty in Reinforcement Learning via Wasserstein Barycenters","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-for-reinforcement-1","slug":"regret-minimization-for-reinforcement-1","title":"Regret Minimization for Reinforcement Learning with Vectorial Feedback and Complex Objectives","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/smile-scalable-meta-inverse-reinforcement","slug":"smile-scalable-meta-inverse-reinforcement","title":"SMILe: Scalable Meta Inverse Reinforcement Learning through Context-Conditional Policies","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/staying-up-to-date-with-online-content","slug":"staying-up-to-date-with-online-content","title":"Staying up to Date with Online Content Changes Using Reinforcement Learning for Scheduling","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/distributed-soft-actor-critic-with","slug":"distributed-soft-actor-critic-with","title":"Distributed Soft Actor-Critic with Multivariate Reward Representation and Knowledge Distillation","date":"2019-11-29","arxiv_id":"1911.13056","repositories_listed":1,"syntology":null},{"url":"/paper/simulation-based-reinforcement-learning-for","slug":"simulation-based-reinforcement-learning-for","title":"Simulation-based reinforcement learning for real-world autonomous driving","date":"2019-11-29","arxiv_id":"1911.12905","repositories_listed":1,"syntology":null},{"url":"/paper/deep-model-based-reinforcement-learning-via","slug":"deep-model-based-reinforcement-learning-via","title":"Deep Model-Based Reinforcement Learning via Estimated Uncertainty and Conservative Policy Optimization","date":"2019-11-28","arxiv_id":"1911.12574","repositories_listed":1,"syntology":null},{"url":"/paper/playing-games-in-the-dark-an-approach-for","slug":"playing-games-in-the-dark-an-approach-for","title":"Playing Games in the Dark: An approach for cross-modality transfer in reinforcement learning","date":"2019-11-28","arxiv_id":"1911.12851","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-games-in-the-dark-an-approach-for#ran","syntology_url":"https://syntology.ai/paper/1911.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12851"}},"official":null}},{"url":"/paper/towards-similarity-graphs-constructed-by-deep","slug":"towards-similarity-graphs-constructed-by-deep","title":"Towards Similarity Graphs Constructed by Deep Reinforcement Learning","date":"2019-11-27","arxiv_id":"1911.12122","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-regularized-offline-reinforcement-1","slug":"behavior-regularized-offline-reinforcement-1","title":"Behavior Regularized Offline Reinforcement Learning","date":"2019-11-26","arxiv_id":"1911.11361","repositories_listed":1,"syntology":null},{"url":"/paper/join-query-optimization-with-deep","slug":"join-query-optimization-with-deep","title":"Join Query Optimization with Deep Reinforcement Learning Algorithms","date":"2019-11-26","arxiv_id":"1911.11689","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-generation-using","slug":"natural-language-generation-using","title":"Natural Language Generation Using Reinforcement Learning with External Rewards","date":"2019-11-26","arxiv_id":"1911.11404","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-model-free-reinforcement-learning","slug":"end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","date":"2019-11-25","arxiv_id":"1911.10868","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-model-free-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.10868"}},"official":{"repos":["valeoai/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-optimize-variational-quantum","slug":"learning-to-optimize-variational-quantum","title":"Learning to Optimize Variational Quantum Circuits to Solve Combinatorial Problems","date":"2019-11-25","arxiv_id":"1911.11071","repositories_listed":1,"syntology":null},{"url":"/paper/corpus-level-end-to-end-exploration-for","slug":"corpus-level-end-to-end-exploration-for","title":"Corpus-Level End-to-End Exploration for Interactive Systems","date":"2019-11-23","arxiv_id":"1912.00753","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-control-of-a-fiber-manufacturing","slug":"dynamic-control-of-a-fiber-manufacturing","title":"Dynamic Control of a Fiber Manufacturing Process using Deep Reinforcement Learning","date":"2019-11-23","arxiv_id":"1911.10286","repositories_listed":1,"syntology":null},{"url":"/paper/deepsynth-program-synthesis-for-automatic","slug":"deepsynth-program-synthesis-for-automatic","title":"DeepSynth: Automata Synthesis for Automatic Task Segmentation in Deep Reinforcement Learning","date":"2019-11-22","arxiv_id":"1911.10244","repositories_listed":1,"syntology":null},{"url":"/paper/fleet-control-using-coregionalized-gaussian","slug":"fleet-control-using-coregionalized-gaussian","title":"Fleet Control using Coregionalized Gaussian Process Policy Iteration","date":"2019-11-22","arxiv_id":"1911.10121","repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-episodic-control","slug":"memory-efficient-episodic-control","title":"Memory-Efficient Episodic Control Reinforcement Learning with Dynamic Online k-means","date":"2019-11-21","arxiv_id":"1911.09560","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with-2","slug":"sample-efficient-reinforcement-learning-with-2","title":"Sample-Efficient Reinforcement Learning with Maximum Entropy Mellowmax Episodic Control","date":"2019-11-21","arxiv_id":"1911.09615","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-curiosity-for-efficient-exploration","slug":"bayesian-curiosity-for-efficient-exploration","title":"Bayesian Curiosity for Efficient Exploration in Reinforcement Learning","date":"2019-11-20","arxiv_id":"1911.08701","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-resource-allocation-in-stream","slug":"generalizable-resource-allocation-in-stream","title":"Generalizable Resource Allocation in Stream Processing via Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08517","repositories_listed":1,"syntology":null},{"url":"/paper/planning-with-goal-conditioned-policies-1","slug":"planning-with-goal-conditioned-policies-1","title":"Planning with Goal-Conditioned Policies","date":"2019-11-19","arxiv_id":"1911.08453","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planning-with-goal-conditioned-policies-1#ran","syntology_url":"https://syntology.ai/paper/1911.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08453"}},"official":null}},{"url":"/paper/influence-aware-memory-for-deep-reinforcement-1","slug":"influence-aware-memory-for-deep-reinforcement-1","title":"Influence-aware Memory Architectures for Deep Reinforcement Learning","date":"2019-11-18","arxiv_id":"1911.07643","repositories_listed":1,"syntology":null},{"url":"/paper/ikea-furniture-assembly-environment-for-long","slug":"ikea-furniture-assembly-environment-for-long","title":"IKEA Furniture Assembly Environment for Long-Horizon Complex Manipulation Tasks","date":"2019-11-17","arxiv_id":"1911.07246","repositories_listed":1,"syntology":null},{"url":"/paper/missingness-as-stability-understanding-the","slug":"missingness-as-stability-understanding-the","title":"Missingness as Stability: Understanding the Structure of Missingness in Longitudinal EHR data and its Impact on Reinforcement Learning in Healthcare","date":"2019-11-16","arxiv_id":"1911.07084","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-market-making-in-a","slug":"reinforcement-learning-for-market-making-in-a","title":"Reinforcement Learning for Market Making in a Multi-agent Dealer Market","date":"2019-11-14","arxiv_id":"1911.05892","repositories_listed":1,"syntology":null},{"url":"/paper/a-convergent-off-policy-temporal-difference","slug":"a-convergent-off-policy-temporal-difference","title":"A Convergent Off-Policy Temporal Difference Algorithm","date":"2019-11-13","arxiv_id":"1911.05697","repositories_listed":1,"syntology":null},{"url":"/paper/drills-deep-reinforcement-learning-for-logic","slug":"drills-deep-reinforcement-learning-for-logic","title":"DRiLLS: Deep Reinforcement Learning for Logic Synthesis","date":"2019-11-11","arxiv_id":"1911.04021","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drills-deep-reinforcement-learning-for-logic#ran","syntology_url":"https://syntology.ai/paper/1911.04021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04021"}},"official":{"repos":["scale-lab/DRiLLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/driving-reinforcement-learning-with-models","slug":"driving-reinforcement-learning-with-models","title":"Driving Reinforcement Learning with Models","date":"2019-11-11","arxiv_id":"1911.04400","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-connected-autonomous-driving","slug":"multi-agent-connected-autonomous-driving","title":"Multi-Agent Connected Autonomous Driving using Deep Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04175","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-connected-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/1911.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04175"}},"official":{"repos":["praveen-palanisamy/macad-gym"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smix-enhancing-centralized-value-functions","slug":"smix-enhancing-centralized-value-functions","title":"SMIX($λ$): Enhancing Centralized Value Functions for Cooperative Multi-Agent Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04094","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-common-question-generation-from","slug":"unsupervised-common-question-generation-from","title":"Contrastive Multi-document Question Generation","date":"2019-11-08","arxiv_id":"1911.03047","repositories_listed":1,"syntology":null},{"url":"/paper/h_inf-model-free-reinforcement-learning-with","slug":"h_inf-model-free-reinforcement-learning-with","title":"$H_\\infty$ Model-free Reinforcement Learning with Robust Stability Guarantee","date":"2019-11-07","arxiv_id":"1911.02875","repositories_listed":1,"syntology":null},{"url":"/paper/improving-reinforcement-learning-algorithms","slug":"improving-reinforcement-learning-algorithms","title":"Improving reinforcement learning algorithms: towards optimal learning rate policies","date":"2019-11-06","arxiv_id":"1911.02319","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-based-approach","slug":"a-deep-reinforcement-learning-based-approach","title":"A Deep Reinforcement Learning Approach to First-Order Logic Theorem Proving","date":"2019-11-05","arxiv_id":"1911.02065","repositories_listed":1,"syntology":null},{"url":"/paper/gym-ignition-reproducible-robotic-simulations","slug":"gym-ignition-reproducible-robotic-simulations","title":"Gym-Ignition: Reproducible Robotic Simulations for Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.01715","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-trajectories-via-subgoal-1","slug":"learning-from-trajectories-via-subgoal-1","title":"Learning from Trajectories via Subgoal Discovery","date":"2019-11-03","arxiv_id":"1911.07224","repositories_listed":1,"syntology":null},{"url":"/paper/on-solving-the-2-dimensional-greedy-shooter","slug":"on-solving-the-2-dimensional-greedy-shooter","title":"On Solving the 2-Dimensional Greedy Shooter Problem for UAVs","date":"2019-11-02","arxiv_id":"1911.01419","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-for-contextual-bandit","slug":"thompson-sampling-for-contextual-bandit","title":"Thompson Sampling for Contextual Bandit Problems with Auxiliary Safety Constraints","date":"2019-11-02","arxiv_id":"1911.00638","repositories_listed":1,"syntology":null},{"url":"/paper/explicit-explore-exploit-algorithms-in","slug":"explicit-explore-exploit-algorithms-in","title":"Explicit Explore-Exploit Algorithms in Continuous State Spaces","date":"2019-11-01","arxiv_id":"1911.00617","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-speedy-q-learning","slug":"generalized-speedy-q-learning","title":"Generalized Speedy Q-learning","date":"2019-11-01","arxiv_id":"1911.00397","repositories_listed":1,"syntology":null},{"url":"/paper/positive-unlabeled-reward-learning","slug":"positive-unlabeled-reward-learning","title":"Positive-Unlabeled Reward Learning","date":"2019-11-01","arxiv_id":"1911.00459","repositories_listed":1,"syntology":null},{"url":"/paper/cascaded-lstms-based-deep-reinforcement","slug":"cascaded-lstms-based-deep-reinforcement","title":"Cascaded LSTMs based Deep Reinforcement Learning for Goal-driven Dialogue","date":"2019-10-31","arxiv_id":"1910.14229","repositories_listed":1,"syntology":null},{"url":"/paper/policy-continuation-with-hindsight-inverse","slug":"policy-continuation-with-hindsight-inverse","title":"Policy Continuation with Hindsight Inverse Dynamics","date":"2019-10-30","arxiv_id":"1910.14055","repositories_listed":1,"syntology":null},{"url":"/paper/191013249","slug":"191013249","title":"Navigation Agents for the Visually Impaired: A Sidewalk Simulator and Experiments","date":"2019-10-29","arxiv_id":"1910.13249","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-deep-energy-based","slug":"a-framework-for-deep-energy-based","title":"Quantum enhancements for deep reinforcement learning in large spaces","date":"2019-10-28","arxiv_id":"1910.12760","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-methods-for-model-based","slug":"asynchronous-methods-for-model-based","title":"Asynchronous Methods for Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12453","repositories_listed":1,"syntology":null},{"url":"/paper/entity-abstraction-in-visual-model-based","slug":"entity-abstraction-in-visual-model-based","title":"Entity Abstraction in Visual Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/entity-abstraction-in-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/1910.12827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12827"}},"official":{"repos":["jcoreyes/OP3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-in-reinforcement-learning-with","slug":"generalization-in-reinforcement-learning-with","title":"Generalization in Reinforcement Learning with Selective Noise Injection and Information Bottleneck","date":"2019-10-28","arxiv_id":"1910.12911","repositories_listed":1,"syntology":null},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","repositories_listed":1,"syntology":null},{"url":"/paper/convergent-policy-optimization-for-safe","slug":"convergent-policy-optimization-for-safe","title":"Convergent Policy Optimization for Safe Reinforcement Learning","date":"2019-10-26","arxiv_id":"1910.12156","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-hol4","slug":"deep-reinforcement-learning-in-hol4","title":"Deep Reinforcement Learning for Synthesizing Functions in Higher-Order Logic","date":"2019-10-25","arxiv_id":"1910.11797","repositories_listed":1,"syntology":null},{"url":"/paper/relay-policy-learning-solving-long-horizon","slug":"relay-policy-learning-solving-long-horizon","title":"Relay Policy Learning: Solving Long-Horizon Tasks via Imitation and Reinforcement Learning","date":"2019-10-25","arxiv_id":"1910.11956","repositories_listed":1,"syntology":null},{"url":"/paper/hrl4in-hierarchical-reinforcement-learning","slug":"hrl4in-hierarchical-reinforcement-learning","title":"HRL4IN: Hierarchical Reinforcement Learning for Interactive Navigation with Mobile Manipulators","date":"2019-10-24","arxiv_id":"1910.11432","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hrl4in-hierarchical-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.11432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11432"}},"official":null}},{"url":"/paper/attention-based-curiosity-driven-exploration","slug":"attention-based-curiosity-driven-exploration","title":"Attention-based Curiosity-driven Exploration in Deep Reinforcement Learning","date":"2019-10-23","arxiv_id":"1910.10840","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-imagined-goals-for-self-supervised","slug":"contextual-imagined-goals-for-self-supervised","title":"Contextual Imagined Goals for Self-Supervised Robotic Learning","date":"2019-10-23","arxiv_id":"1910.11670","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-imagined-goals-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/1910.11670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11670"}},"official":null}},{"url":"/paper/efficient-decoupled-neural-architecture","slug":"efficient-decoupled-neural-architecture","title":"Efficient Decoupled Neural Architecture Search by Structure and Operation Sampling","date":"2019-10-23","arxiv_id":"1910.10397","repositories_listed":1,"syntology":null},{"url":"/paper/teach-biped-robots-to-walk-via-gait","slug":"teach-biped-robots-to-walk-via-gait","title":"Teach Biped Robots to Walk via Gait Principles and Reinforcement Learning with Adversarial Critics","date":"2019-10-22","arxiv_id":"1910.10194","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-skill-networks-unsupervised-robot","slug":"adversarial-skill-networks-unsupervised-robot","title":"Adversarial Skill Networks: Unsupervised Robot Skill Learning from Video","date":"2019-10-21","arxiv_id":"1910.09430","repositories_listed":1,"syntology":null},{"url":"/paper/dealing-with-sparse-rewards-in-reinforcement","slug":"dealing-with-sparse-rewards-in-reinforcement","title":"Dealing with Sparse Rewards in Reinforcement Learning","date":"2019-10-21","arxiv_id":"1910.09281","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-control-of","slug":"deep-reinforcement-learning-control-of","title":"Deep Reinforcement Learning Control of Quantum Cartpoles","date":"2019-10-21","arxiv_id":"1910.09200","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-map-natural-language-instructions","slug":"learning-to-map-natural-language-instructions","title":"Learning to Map Natural Language Instructions to Physical Quadcopter Control using Simulated Flight","date":"2019-10-21","arxiv_id":"1910.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-map-natural-language-instructions#ran","syntology_url":"https://syntology.ai/paper/1910.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09664"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rlscheduler-learn-to-schedule-hpc-batch-jobs","slug":"rlscheduler-learn-to-schedule-hpc-batch-jobs","title":"RLScheduler: An Automated HPC Batch Job Scheduler Using Reinforcement Learning","date":"2019-10-20","arxiv_id":"1910.08925","repositories_listed":1,"syntology":null},{"url":"/paper/a-structured-prediction-approach-for","slug":"a-structured-prediction-approach-for","title":"A Structured Prediction Approach for Generalization in Cooperative Multi-Agent Reinforcement Learning","date":"2019-10-19","arxiv_id":"1910.08809","repositories_listed":1,"syntology":null},{"url":"/paper/natural-question-generation-with","slug":"natural-question-generation-with","title":"Natural Question Generation with Reinforcement Learning Based Graph-to-Sequence Model","date":"2019-10-19","arxiv_id":"1910.08832","repositories_listed":1,"syntology":null},{"url":"/paper/towards-more-sample-efficiency","slug":"towards-more-sample-efficiency","title":"Towards More Sample Efficiency in Reinforcement Learning with Data Augmentation","date":"2019-10-19","arxiv_id":"1910.09959","repositories_listed":1,"syntology":null},{"url":"/paper/multi-view-reinforcement-learning","slug":"multi-view-reinforcement-learning","title":"Multi-View Reinforcement Learning","date":"2019-10-18","arxiv_id":"1910.08285","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-curriculum-generation-from","slug":"adaptive-curriculum-generation-from","title":"Adaptive Curriculum Generation from Demonstrations for Sim-to-Real Visuomotor Control","date":"2019-10-17","arxiv_id":"1910.07972","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-discretization-for-episodic","slug":"adaptive-discretization-for-episodic","title":"Adaptive Discretization for Episodic Reinforcement Learning in Metric Spaces","date":"2019-10-17","arxiv_id":"1910.08151","repositories_listed":1,"syntology":null},{"url":"/paper/single-episode-policy-transfer-in","slug":"single-episode-policy-transfer-in","title":"Single Episode Policy Transfer in Reinforcement Learning","date":"2019-10-17","arxiv_id":"1910.07719","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/single-episode-policy-transfer-in#ran","syntology_url":"https://syntology.ai/paper/1910.07719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07719"}},"official":{"repos":["011235813/SEPT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-meets-graph","slug":"deep-reinforcement-learning-meets-graph","title":"Deep Reinforcement Learning meets Graph Neural Networks: exploring a routing optimization use case","date":"2019-10-16","arxiv_id":"1910.07421","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-meets-graph#ran","syntology_url":"https://syntology.ai/paper/1910.07421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07421"}},"official":{"repos":["knowledgedefinednetworking/DRL-GNN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/model-free-reinforcement-learning-in-infinite","slug":"model-free-reinforcement-learning-in-infinite","title":"Model-free Reinforcement Learning in Infinite-horizon Average-reward Markov Decision Processes","date":"2019-10-15","arxiv_id":"1910.07072","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-spiking-coagents","slug":"reinforcement-learning-with-spiking-coagents","title":"Reinforcement learning with a network of spiking agents","date":"2019-10-15","arxiv_id":"1910.06489","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapping-the-expressivity-with-model","slug":"bootstrapping-the-expressivity-with-model","title":"On the Expressivity of Neural Networks for Deep Reinforcement Learning","date":"2019-10-14","arxiv_id":"1910.05927","repositories_listed":1,"syntology":null},{"url":"/paper/policy-poisoning-in-batch-reinforcement","slug":"policy-poisoning-in-batch-reinforcement","title":"Policy Poisoning in Batch Reinforcement Learning and Control","date":"2019-10-13","arxiv_id":"1910.05821","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-navigation-via-deep-reinforcement","slug":"autonomous-navigation-via-deep-reinforcement","title":"Autonomous Navigation via Deep Reinforcement Learning for Resource Constraint Edge Nodes using Transfer Learning","date":"2019-10-12","arxiv_id":"1910.05547","repositories_listed":1,"syntology":null},{"url":"/paper/influence-based-multi-agent-exploration","slug":"influence-based-multi-agent-exploration","title":"Influence-Based Multi-Agent Exploration","date":"2019-10-12","arxiv_id":"1910.05512","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/influence-based-multi-agent-exploration#ran","syntology_url":"https://syntology.ai/paper/1910.05512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05512"}},"official":{"repos":["TonghanWang/EITI-EDTI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-with-warm-start-and-active-termination","slug":"agent-with-warm-start-and-active-termination","title":"Agent with Warm Start and Active Termination for Plane Localization in 3D Ultrasound","date":"2019-10-10","arxiv_id":"1910.04331","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-with-3","slug":"hierarchical-reinforcement-learning-with-3","title":"Hierarchical Reinforcement Learning with Advantage-Based Auxiliary Rewards","date":"2019-10-10","arxiv_id":"1910.04450","repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-contextual-reinforcement-learning","slug":"self-paced-contextual-reinforcement-learning","title":"Self-Paced Contextual Reinforcement Learning","date":"2019-10-07","arxiv_id":"1910.02826","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-paced-contextual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.02826"}},"official":{"repos":["psclklnk/self-paced-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"ca0b6ecb5fd97c9382fd55d78941604e46e547e660be8d0b8f5832f8e1e45046","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}