{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/36","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":36,"pages_in_order":135,"rows_per_page":100,"rows":[3501,3600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/35","next":"/task/reinforcement-learning-2/papers/37","papers":[{"url":"/paper/pixelrl-fully-convolutional-network-with","slug":"pixelrl-fully-convolutional-network-with","title":"PixelRL: Fully Convolutional Network with Reinforcement Learning for Image Processing","date":"2019-12-16","arxiv_id":"1912.07190","repositories_listed":1,"syntology":null},{"url":"/paper/unas-differentiable-architecture-search-meets","slug":"unas-differentiable-architecture-search-meets","title":"UNAS: Differentiable Architecture Search Meets Reinforcement Learning","date":"2019-12-16","arxiv_id":"1912.07651","repositories_listed":1,"syntology":null},{"url":"/paper/dota-2-with-large-scale-deep-reinforcement","slug":"dota-2-with-large-scale-deep-reinforcement","title":"Dota 2 with Large Scale Deep Reinforcement Learning","date":"2019-12-13","arxiv_id":"1912.06680","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dota-2-with-large-scale-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1912.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.06680"}},"official":null}},{"url":"/paper/the-playstation-reinforcement-learning","slug":"the-playstation-reinforcement-learning","title":"The PlayStation Reinforcement Learning Environment (PSXLE)","date":"2019-12-12","arxiv_id":"1912.06101","repositories_listed":1,"syntology":null},{"url":"/paper/smirl-surprise-minimizing-rl-in-dynamic","slug":"smirl-surprise-minimizing-rl-in-dynamic","title":"SMiRL: Surprise Minimizing Reinforcement Learning in Unstable Environments","date":"2019-12-11","arxiv_id":"1912.05510","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-reliability-of-reinforcement-1","slug":"measuring-the-reliability-of-reinforcement-1","title":"Measuring the Reliability of Reinforcement Learning Algorithms","date":"2019-12-10","arxiv_id":"1912.05663","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-the-reliability-of-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1912.05663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05663"}},"official":{"repos":["google-research/rl-reliability-metrics"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chainerrl-a-deep-reinforcement-learning","slug":"chainerrl-a-deep-reinforcement-learning","title":"ChainerRL: A Deep Reinforcement Learning Library","date":"2019-12-09","arxiv_id":"1912.03905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chainerrl-a-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1912.03905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03905"}},"official":{"repos":["chainer/chainerrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploratory-not-explanatory-counterfactual-1","slug":"exploratory-not-explanatory-counterfactual-1","title":"Exploratory Not Explanatory: Counterfactual Analysis of Saliency Maps for Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.05743","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-cooperative-multi-agent","slug":"hierarchical-cooperative-multi-agent","title":"Hierarchical Cooperative Multi-Agent Reinforcement Learning with Skill Discovery","date":"2019-12-07","arxiv_id":"1912.03558","repositories_listed":1,"syntology":null},{"url":"/paper/increasing-performance-of-electric-vehicles","slug":"increasing-performance-of-electric-vehicles","title":"Increasing performance of electric vehicles in ride-hailing services using deep reinforcement learning","date":"2019-12-07","arxiv_id":"1912.03408","repositories_listed":1,"syntology":null},{"url":"/paper/valan-vision-and-language-agent-navigation","slug":"valan-vision-and-language-agent-navigation","title":"VALAN: Vision and Language Agent Navigation","date":"2019-12-06","arxiv_id":"1912.03241","repositories_listed":1,"syntology":null},{"url":"/paper/191202368","slug":"191202368","title":"Inter-Level Cooperation in Hierarchical Reinforcement Learning","date":"2019-12-05","arxiv_id":"1912.02368","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-credit-assignment-1","slug":"hindsight-credit-assignment-1","title":"Hindsight Credit Assignment","date":"2019-12-05","arxiv_id":"1912.02503","repositories_listed":1,"syntology":null},{"url":"/paper/mo-states-mo-problems-emergency-stop-1","slug":"mo-states-mo-problems-emergency-stop-1","title":"Mo' States Mo' Problems: Emergency Stop Mechanisms from Observation","date":"2019-12-03","arxiv_id":"1912.01649","repositories_listed":1,"syntology":null},{"url":"/paper/safelife-10-exploring-side-effects-in-complex","slug":"safelife-10-exploring-side-effects-in-complex","title":"SafeLife 1.0: Exploring Side Effects in Complex Environments","date":"2019-12-03","arxiv_id":"1912.01217","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safelife-10-exploring-side-effects-in-complex#ran","syntology_url":"https://syntology.ai/paper/1912.01217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01217"}},"official":{"repos":["PartnershipOnAI/safelife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-model-based-reinforcement-learning-with-1","slug":"a-model-based-reinforcement-learning-with-1","title":"A Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-auxiliary-task-weighting-for","slug":"adaptive-auxiliary-task-weighting-for","title":"Adaptive Auxiliary Task Weighting for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-local-search-heuristics-for-boolean","slug":"learning-local-search-heuristics-for-boolean","title":"Learning Local Search Heuristics for Boolean Satisfiability","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-reward-machines-for-partially","slug":"learning-reward-machines-for-partially","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/liir-learning-individual-intrinsic-reward-in","slug":"liir-learning-individual-intrinsic-reward-in","title":"LIIR: Learning Individual Intrinsic Reward in Multi-Agent Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/loaded-dice-trading-off-bias-and-variance-in-1","slug":"loaded-dice-trading-off-bias-and-variance-in-1","title":"Loaded DiCE: Trading off Bias and Variance in Any-Order Score Function Gradient Estimators for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/park-an-open-platform-for-learning-augmented","slug":"park-an-open-platform-for-learning-augmented","title":"Park: An Open Platform for Learning-Augmented Computer Systems","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/privacy-preserving-q-learning-with-functional","slug":"privacy-preserving-q-learning-with-functional","title":"Privacy-Preserving Q-Learning with Functional Noise in Continuous Spaces","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/propagating-uncertainty-in-reinforcement","slug":"propagating-uncertainty-in-reinforcement","title":"Propagating Uncertainty in Reinforcement Learning via Wasserstein Barycenters","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-for-reinforcement-1","slug":"regret-minimization-for-reinforcement-1","title":"Regret Minimization for Reinforcement Learning with Vectorial Feedback and Complex Objectives","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/smile-scalable-meta-inverse-reinforcement","slug":"smile-scalable-meta-inverse-reinforcement","title":"SMILe: Scalable Meta Inverse Reinforcement Learning through Context-Conditional Policies","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/staying-up-to-date-with-online-content","slug":"staying-up-to-date-with-online-content","title":"Staying up to Date with Online Content Changes Using Reinforcement Learning for Scheduling","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/distributed-soft-actor-critic-with","slug":"distributed-soft-actor-critic-with","title":"Distributed Soft Actor-Critic with Multivariate Reward Representation and Knowledge Distillation","date":"2019-11-29","arxiv_id":"1911.13056","repositories_listed":1,"syntology":null},{"url":"/paper/simulation-based-reinforcement-learning-for","slug":"simulation-based-reinforcement-learning-for","title":"Simulation-based reinforcement learning for real-world autonomous driving","date":"2019-11-29","arxiv_id":"1911.12905","repositories_listed":1,"syntology":null},{"url":"/paper/deep-model-based-reinforcement-learning-via","slug":"deep-model-based-reinforcement-learning-via","title":"Deep Model-Based Reinforcement Learning via Estimated Uncertainty and Conservative Policy Optimization","date":"2019-11-28","arxiv_id":"1911.12574","repositories_listed":1,"syntology":null},{"url":"/paper/playing-games-in-the-dark-an-approach-for","slug":"playing-games-in-the-dark-an-approach-for","title":"Playing Games in the Dark: An approach for cross-modality transfer in reinforcement learning","date":"2019-11-28","arxiv_id":"1911.12851","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-games-in-the-dark-an-approach-for#ran","syntology_url":"https://syntology.ai/paper/1911.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12851"}},"official":null}},{"url":"/paper/towards-similarity-graphs-constructed-by-deep","slug":"towards-similarity-graphs-constructed-by-deep","title":"Towards Similarity Graphs Constructed by Deep Reinforcement Learning","date":"2019-11-27","arxiv_id":"1911.12122","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-regularized-offline-reinforcement-1","slug":"behavior-regularized-offline-reinforcement-1","title":"Behavior Regularized Offline Reinforcement Learning","date":"2019-11-26","arxiv_id":"1911.11361","repositories_listed":1,"syntology":null},{"url":"/paper/join-query-optimization-with-deep","slug":"join-query-optimization-with-deep","title":"Join Query Optimization with Deep Reinforcement Learning Algorithms","date":"2019-11-26","arxiv_id":"1911.11689","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-generation-using","slug":"natural-language-generation-using","title":"Natural Language Generation Using Reinforcement Learning with External Rewards","date":"2019-11-26","arxiv_id":"1911.11404","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-model-free-reinforcement-learning","slug":"end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","date":"2019-11-25","arxiv_id":"1911.10868","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-model-free-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.10868"}},"official":{"repos":["valeoai/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-control-of-a-fiber-manufacturing","slug":"dynamic-control-of-a-fiber-manufacturing","title":"Dynamic Control of a Fiber Manufacturing Process using Deep Reinforcement Learning","date":"2019-11-23","arxiv_id":"1911.10286","repositories_listed":1,"syntology":null},{"url":"/paper/deepsynth-program-synthesis-for-automatic","slug":"deepsynth-program-synthesis-for-automatic","title":"DeepSynth: Automata Synthesis for Automatic Task Segmentation in Deep Reinforcement Learning","date":"2019-11-22","arxiv_id":"1911.10244","repositories_listed":1,"syntology":null},{"url":"/paper/fleet-control-using-coregionalized-gaussian","slug":"fleet-control-using-coregionalized-gaussian","title":"Fleet Control using Coregionalized Gaussian Process Policy Iteration","date":"2019-11-22","arxiv_id":"1911.10121","repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-episodic-control","slug":"memory-efficient-episodic-control","title":"Memory-Efficient Episodic Control Reinforcement Learning with Dynamic Online k-means","date":"2019-11-21","arxiv_id":"1911.09560","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with-2","slug":"sample-efficient-reinforcement-learning-with-2","title":"Sample-Efficient Reinforcement Learning with Maximum Entropy Mellowmax Episodic Control","date":"2019-11-21","arxiv_id":"1911.09615","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-curiosity-for-efficient-exploration","slug":"bayesian-curiosity-for-efficient-exploration","title":"Bayesian Curiosity for Efficient Exploration in Reinforcement Learning","date":"2019-11-20","arxiv_id":"1911.08701","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-resource-allocation-in-stream","slug":"generalizable-resource-allocation-in-stream","title":"Generalizable Resource Allocation in Stream Processing via Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08517","repositories_listed":1,"syntology":null},{"url":"/paper/planning-with-goal-conditioned-policies-1","slug":"planning-with-goal-conditioned-policies-1","title":"Planning with Goal-Conditioned Policies","date":"2019-11-19","arxiv_id":"1911.08453","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planning-with-goal-conditioned-policies-1#ran","syntology_url":"https://syntology.ai/paper/1911.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08453"}},"official":null}},{"url":"/paper/influence-aware-memory-for-deep-reinforcement-1","slug":"influence-aware-memory-for-deep-reinforcement-1","title":"Influence-aware Memory Architectures for Deep Reinforcement Learning","date":"2019-11-18","arxiv_id":"1911.07643","repositories_listed":1,"syntology":null},{"url":"/paper/ikea-furniture-assembly-environment-for-long","slug":"ikea-furniture-assembly-environment-for-long","title":"IKEA Furniture Assembly Environment for Long-Horizon Complex Manipulation Tasks","date":"2019-11-17","arxiv_id":"1911.07246","repositories_listed":1,"syntology":null},{"url":"/paper/missingness-as-stability-understanding-the","slug":"missingness-as-stability-understanding-the","title":"Missingness as Stability: Understanding the Structure of Missingness in Longitudinal EHR data and its Impact on Reinforcement Learning in Healthcare","date":"2019-11-16","arxiv_id":"1911.07084","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-market-making-in-a","slug":"reinforcement-learning-for-market-making-in-a","title":"Reinforcement Learning for Market Making in a Multi-agent Dealer Market","date":"2019-11-14","arxiv_id":"1911.05892","repositories_listed":1,"syntology":null},{"url":"/paper/drills-deep-reinforcement-learning-for-logic","slug":"drills-deep-reinforcement-learning-for-logic","title":"DRiLLS: Deep Reinforcement Learning for Logic Synthesis","date":"2019-11-11","arxiv_id":"1911.04021","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drills-deep-reinforcement-learning-for-logic#ran","syntology_url":"https://syntology.ai/paper/1911.04021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04021"}},"official":{"repos":["scale-lab/DRiLLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/driving-reinforcement-learning-with-models","slug":"driving-reinforcement-learning-with-models","title":"Driving Reinforcement Learning with Models","date":"2019-11-11","arxiv_id":"1911.04400","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-connected-autonomous-driving","slug":"multi-agent-connected-autonomous-driving","title":"Multi-Agent Connected Autonomous Driving using Deep Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04175","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-connected-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/1911.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04175"}},"official":{"repos":["praveen-palanisamy/macad-gym"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smix-enhancing-centralized-value-functions","slug":"smix-enhancing-centralized-value-functions","title":"SMIX($λ$): Enhancing Centralized Value Functions for Cooperative Multi-Agent Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04094","repositories_listed":1,"syntology":null},{"url":"/paper/h_inf-model-free-reinforcement-learning-with","slug":"h_inf-model-free-reinforcement-learning-with","title":"$H_\\infty$ Model-free Reinforcement Learning with Robust Stability Guarantee","date":"2019-11-07","arxiv_id":"1911.02875","repositories_listed":1,"syntology":null},{"url":"/paper/improving-reinforcement-learning-algorithms","slug":"improving-reinforcement-learning-algorithms","title":"Improving reinforcement learning algorithms: towards optimal learning rate policies","date":"2019-11-06","arxiv_id":"1911.02319","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-based-approach","slug":"a-deep-reinforcement-learning-based-approach","title":"A Deep Reinforcement Learning Approach to First-Order Logic Theorem Proving","date":"2019-11-05","arxiv_id":"1911.02065","repositories_listed":1,"syntology":null},{"url":"/paper/gym-ignition-reproducible-robotic-simulations","slug":"gym-ignition-reproducible-robotic-simulations","title":"Gym-Ignition: Reproducible Robotic Simulations for Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.01715","repositories_listed":1,"syntology":null},{"url":"/paper/on-solving-the-2-dimensional-greedy-shooter","slug":"on-solving-the-2-dimensional-greedy-shooter","title":"On Solving the 2-Dimensional Greedy Shooter Problem for UAVs","date":"2019-11-02","arxiv_id":"1911.01419","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-for-contextual-bandit","slug":"thompson-sampling-for-contextual-bandit","title":"Thompson Sampling for Contextual Bandit Problems with Auxiliary Safety Constraints","date":"2019-11-02","arxiv_id":"1911.00638","repositories_listed":1,"syntology":null},{"url":"/paper/cascaded-lstms-based-deep-reinforcement","slug":"cascaded-lstms-based-deep-reinforcement","title":"Cascaded LSTMs based Deep Reinforcement Learning for Goal-driven Dialogue","date":"2019-10-31","arxiv_id":"1910.14229","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-deep-energy-based","slug":"a-framework-for-deep-energy-based","title":"Quantum enhancements for deep reinforcement learning in large spaces","date":"2019-10-28","arxiv_id":"1910.12760","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-methods-for-model-based","slug":"asynchronous-methods-for-model-based","title":"Asynchronous Methods for Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12453","repositories_listed":1,"syntology":null},{"url":"/paper/entity-abstraction-in-visual-model-based","slug":"entity-abstraction-in-visual-model-based","title":"Entity Abstraction in Visual Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/entity-abstraction-in-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/1910.12827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12827"}},"official":{"repos":["jcoreyes/OP3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-in-reinforcement-learning-with","slug":"generalization-in-reinforcement-learning-with","title":"Generalization in Reinforcement Learning with Selective Noise Injection and Information Bottleneck","date":"2019-10-28","arxiv_id":"1910.12911","repositories_listed":1,"syntology":null},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","repositories_listed":1,"syntology":null},{"url":"/paper/convergent-policy-optimization-for-safe","slug":"convergent-policy-optimization-for-safe","title":"Convergent Policy Optimization for Safe Reinforcement Learning","date":"2019-10-26","arxiv_id":"1910.12156","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-hol4","slug":"deep-reinforcement-learning-in-hol4","title":"Deep Reinforcement Learning for Synthesizing Functions in Higher-Order Logic","date":"2019-10-25","arxiv_id":"1910.11797","repositories_listed":1,"syntology":null},{"url":"/paper/relay-policy-learning-solving-long-horizon","slug":"relay-policy-learning-solving-long-horizon","title":"Relay Policy Learning: Solving Long-Horizon Tasks via Imitation and Reinforcement Learning","date":"2019-10-25","arxiv_id":"1910.11956","repositories_listed":1,"syntology":null},{"url":"/paper/hrl4in-hierarchical-reinforcement-learning","slug":"hrl4in-hierarchical-reinforcement-learning","title":"HRL4IN: Hierarchical Reinforcement Learning for Interactive Navigation with Mobile Manipulators","date":"2019-10-24","arxiv_id":"1910.11432","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hrl4in-hierarchical-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.11432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11432"}},"official":null}},{"url":"/paper/attention-based-curiosity-driven-exploration","slug":"attention-based-curiosity-driven-exploration","title":"Attention-based Curiosity-driven Exploration in Deep Reinforcement Learning","date":"2019-10-23","arxiv_id":"1910.10840","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-imagined-goals-for-self-supervised","slug":"contextual-imagined-goals-for-self-supervised","title":"Contextual Imagined Goals for Self-Supervised Robotic Learning","date":"2019-10-23","arxiv_id":"1910.11670","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-imagined-goals-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/1910.11670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11670"}},"official":null}},{"url":"/paper/efficient-decoupled-neural-architecture","slug":"efficient-decoupled-neural-architecture","title":"Efficient Decoupled Neural Architecture Search by Structure and Operation Sampling","date":"2019-10-23","arxiv_id":"1910.10397","repositories_listed":1,"syntology":null},{"url":"/paper/dealing-with-sparse-rewards-in-reinforcement","slug":"dealing-with-sparse-rewards-in-reinforcement","title":"Dealing with Sparse Rewards in Reinforcement Learning","date":"2019-10-21","arxiv_id":"1910.09281","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-control-of","slug":"deep-reinforcement-learning-control-of","title":"Deep Reinforcement Learning Control of Quantum Cartpoles","date":"2019-10-21","arxiv_id":"1910.09200","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-map-natural-language-instructions","slug":"learning-to-map-natural-language-instructions","title":"Learning to Map Natural Language Instructions to Physical Quadcopter Control using Simulated Flight","date":"2019-10-21","arxiv_id":"1910.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-map-natural-language-instructions#ran","syntology_url":"https://syntology.ai/paper/1910.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09664"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rlscheduler-learn-to-schedule-hpc-batch-jobs","slug":"rlscheduler-learn-to-schedule-hpc-batch-jobs","title":"RLScheduler: An Automated HPC Batch Job Scheduler Using Reinforcement Learning","date":"2019-10-20","arxiv_id":"1910.08925","repositories_listed":1,"syntology":null},{"url":"/paper/a-structured-prediction-approach-for","slug":"a-structured-prediction-approach-for","title":"A Structured Prediction Approach for Generalization in Cooperative Multi-Agent Reinforcement Learning","date":"2019-10-19","arxiv_id":"1910.08809","repositories_listed":1,"syntology":null},{"url":"/paper/natural-question-generation-with","slug":"natural-question-generation-with","title":"Natural Question Generation with Reinforcement Learning Based Graph-to-Sequence Model","date":"2019-10-19","arxiv_id":"1910.08832","repositories_listed":1,"syntology":null},{"url":"/paper/towards-more-sample-efficiency","slug":"towards-more-sample-efficiency","title":"Towards More Sample Efficiency in Reinforcement Learning with Data Augmentation","date":"2019-10-19","arxiv_id":"1910.09959","repositories_listed":1,"syntology":null},{"url":"/paper/multi-view-reinforcement-learning","slug":"multi-view-reinforcement-learning","title":"Multi-View Reinforcement Learning","date":"2019-10-18","arxiv_id":"1910.08285","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-discretization-for-episodic","slug":"adaptive-discretization-for-episodic","title":"Adaptive Discretization for Episodic Reinforcement Learning in Metric Spaces","date":"2019-10-17","arxiv_id":"1910.08151","repositories_listed":1,"syntology":null},{"url":"/paper/single-episode-policy-transfer-in","slug":"single-episode-policy-transfer-in","title":"Single Episode Policy Transfer in Reinforcement Learning","date":"2019-10-17","arxiv_id":"1910.07719","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/single-episode-policy-transfer-in#ran","syntology_url":"https://syntology.ai/paper/1910.07719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07719"}},"official":{"repos":["011235813/SEPT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/model-free-reinforcement-learning-in-infinite","slug":"model-free-reinforcement-learning-in-infinite","title":"Model-free Reinforcement Learning in Infinite-horizon Average-reward Markov Decision Processes","date":"2019-10-15","arxiv_id":"1910.07072","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-spiking-coagents","slug":"reinforcement-learning-with-spiking-coagents","title":"Reinforcement learning with a network of spiking agents","date":"2019-10-15","arxiv_id":"1910.06489","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapping-the-expressivity-with-model","slug":"bootstrapping-the-expressivity-with-model","title":"On the Expressivity of Neural Networks for Deep Reinforcement Learning","date":"2019-10-14","arxiv_id":"1910.05927","repositories_listed":1,"syntology":null},{"url":"/paper/policy-poisoning-in-batch-reinforcement","slug":"policy-poisoning-in-batch-reinforcement","title":"Policy Poisoning in Batch Reinforcement Learning and Control","date":"2019-10-13","arxiv_id":"1910.05821","repositories_listed":1,"syntology":null},{"url":"/paper/influence-based-multi-agent-exploration","slug":"influence-based-multi-agent-exploration","title":"Influence-Based Multi-Agent Exploration","date":"2019-10-12","arxiv_id":"1910.05512","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/influence-based-multi-agent-exploration#ran","syntology_url":"https://syntology.ai/paper/1910.05512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05512"}},"official":{"repos":["TonghanWang/EITI-EDTI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-reinforcement-learning-with-3","slug":"hierarchical-reinforcement-learning-with-3","title":"Hierarchical Reinforcement Learning with Advantage-Based Auxiliary Rewards","date":"2019-10-10","arxiv_id":"1910.04450","repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-contextual-reinforcement-learning","slug":"self-paced-contextual-reinforcement-learning","title":"Self-Paced Contextual Reinforcement Learning","date":"2019-10-07","arxiv_id":"1910.02826","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-paced-contextual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.02826"}},"official":{"repos":["psclklnk/self-paced-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantized-reinforcement-learning-quarl#ran","syntology_url":"https://syntology.ai/paper/1910.01055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01055"}},"official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiagent-rollout-algorithms-and","slug":"multiagent-rollout-algorithms-and","title":"Multiagent Rollout Algorithms and Reinforcement Learning","date":"2019-09-30","arxiv_id":"1910.00120","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-roi-generation-for-video-object","slug":"adaptive-roi-generation-for-video-object","title":"Adaptive ROI Generation for Video Object Segmentation Using Reinforcement Learning","date":"2019-09-27","arxiv_id":"1909.12482","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-data-driven-robotics","slug":"a-framework-for-data-driven-robotics","title":"Scaling data-driven robotics with reward sketching and batch reinforcement learning","date":"2019-09-26","arxiv_id":"1909.12200","repositories_listed":1,"syntology":null},{"url":"/paper/a-simulation-of-uav-power-optimization-via","slug":"a-simulation-of-uav-power-optimization-via","title":"Visual Exploration and Energy-aware Path Planning via Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12217","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-structures-for-value-based","slug":"harnessing-structures-for-value-based","title":"Harnessing Structures for Value-Based Planning and Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12255","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-structures-for-value-based#ran","syntology_url":"https://syntology.ai/paper/1909.12255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12255"}},"official":{"repos":["YyzHarry/SV-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/good-robot-efficient-reinforcement-learning","slug":"good-robot-efficient-reinforcement-learning","title":"\"Good Robot!\": Efficient Reinforcement Learning for Multi-Step Visual Tasks with Sim to Real Transfer","date":"2019-09-25","arxiv_id":"1909.11730","repositories_listed":1,"syntology":null},{"url":"/paper/robel-robotics-benchmarks-for-learning-with","slug":"robel-robotics-benchmarks-for-learning-with","title":"ROBEL: Robotics Benchmarks for Learning with Low-Cost Robots","date":"2019-09-25","arxiv_id":"1909.11639","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-state-control-through","slug":"self-supervised-state-control-through","title":"Self-Supervised State-Control through Intrinsic Mutual Information Rewards","date":"2019-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/avoidance-learning-using-observational","slug":"avoidance-learning-using-observational","title":"Avoidance Learning Using Observational Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.11228","repositories_listed":1,"syntology":null},{"url":"/paper/invariant-transform-experience-replay","slug":"invariant-transform-experience-replay","title":"Invariant Transform Experience Replay: Data Augmentation for Deep Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.10707","repositories_listed":1,"syntology":null}],"record_sha256":"d4331c37b38778e25459372bd3b196db3db08531b9e206d4bc10f4590a6c0a0a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}