{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/33","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":33,"pages_in_order":132,"rows_per_page":100,"rows":[3201,3300],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/32","next":"/task/reinforcement-learning/papers/34","papers":[{"url":"/paper/random-path-selection-for-continual-learning","slug":"random-path-selection-for-continual-learning","title":"Random Path Selection for Continual Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reconciling-returns-with-experience-replay","slug":"reconciling-returns-with-experience-replay","title":"Reconciling λ-Returns with Experience Replay","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-for-reinforcement-1","slug":"regret-minimization-for-reinforcement-1","title":"Regret Minimization for Reinforcement Learning with Vectorial Feedback and Complex Objectives","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/smile-scalable-meta-inverse-reinforcement","slug":"smile-scalable-meta-inverse-reinforcement","title":"SMILe: Scalable Meta Inverse Reinforcement Learning through Context-Conditional Policies","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/staying-up-to-date-with-online-content","slug":"staying-up-to-date-with-online-content","title":"Staying up to Date with Online Content Changes Using Reinforcement Learning for Scheduling","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/distributed-soft-actor-critic-with","slug":"distributed-soft-actor-critic-with","title":"Distributed Soft Actor-Critic with Multivariate Reward Representation and Knowledge Distillation","date":"2019-11-29","arxiv_id":"1911.13056","repositories_listed":1,"syntology":null},{"url":"/paper/simulation-based-reinforcement-learning-for","slug":"simulation-based-reinforcement-learning-for","title":"Simulation-based reinforcement learning for real-world autonomous driving","date":"2019-11-29","arxiv_id":"1911.12905","repositories_listed":1,"syntology":null},{"url":"/paper/deep-model-based-reinforcement-learning-via","slug":"deep-model-based-reinforcement-learning-via","title":"Deep Model-Based Reinforcement Learning via Estimated Uncertainty and Conservative Policy Optimization","date":"2019-11-28","arxiv_id":"1911.12574","repositories_listed":1,"syntology":null},{"url":"/paper/playing-games-in-the-dark-an-approach-for","slug":"playing-games-in-the-dark-an-approach-for","title":"Playing Games in the Dark: An approach for cross-modality transfer in reinforcement learning","date":"2019-11-28","arxiv_id":"1911.12851","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-games-in-the-dark-an-approach-for#ran","syntology_url":"https://syntology.ai/paper/1911.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12851"}},"official":null}},{"url":"/paper/towards-similarity-graphs-constructed-by-deep","slug":"towards-similarity-graphs-constructed-by-deep","title":"Towards Similarity Graphs Constructed by Deep Reinforcement Learning","date":"2019-11-27","arxiv_id":"1911.12122","repositories_listed":1,"syntology":null},{"url":"/paper/an-autonomous-spectrum-management-scheme-for","slug":"an-autonomous-spectrum-management-scheme-for","title":"An Autonomous Spectrum Management Scheme for Unmanned Aerial Vehicle Networks in Disaster Relief Operations","date":"2019-11-26","arxiv_id":"1911.11343","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-regularized-offline-reinforcement-1","slug":"behavior-regularized-offline-reinforcement-1","title":"Behavior Regularized Offline Reinforcement Learning","date":"2019-11-26","arxiv_id":"1911.11361","repositories_listed":1,"syntology":null},{"url":"/paper/join-query-optimization-with-deep","slug":"join-query-optimization-with-deep","title":"Join Query Optimization with Deep Reinforcement Learning Algorithms","date":"2019-11-26","arxiv_id":"1911.11689","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-generation-using","slug":"natural-language-generation-using","title":"Natural Language Generation Using Reinforcement Learning with External Rewards","date":"2019-11-26","arxiv_id":"1911.11404","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-model-free-reinforcement-learning","slug":"end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","date":"2019-11-25","arxiv_id":"1911.10868","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-model-free-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.10868"}},"official":{"repos":["valeoai/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-optimize-variational-quantum","slug":"learning-to-optimize-variational-quantum","title":"Learning to Optimize Variational Quantum Circuits to Solve Combinatorial Problems","date":"2019-11-25","arxiv_id":"1911.11071","repositories_listed":1,"syntology":null},{"url":"/paper/combined-model-for-partially-observable-and","slug":"combined-model-for-partially-observable-and","title":"Combined Model for Partially-Observable and Non-Observable Task Switching: Solving Hierarchical Reinforcement Learning Problems Statically and Dynamically with Transfer Learning","date":"2019-11-23","arxiv_id":"1911.10425","repositories_listed":1,"syntology":null},{"url":"/paper/corpus-level-end-to-end-exploration-for","slug":"corpus-level-end-to-end-exploration-for","title":"Corpus-Level End-to-End Exploration for Interactive Systems","date":"2019-11-23","arxiv_id":"1912.00753","repositories_listed":1,"syntology":null},{"url":"/paper/deepsynth-program-synthesis-for-automatic","slug":"deepsynth-program-synthesis-for-automatic","title":"DeepSynth: Automata Synthesis for Automatic Task Segmentation in Deep Reinforcement Learning","date":"2019-11-22","arxiv_id":"1911.10244","repositories_listed":1,"syntology":null},{"url":"/paper/fleet-control-using-coregionalized-gaussian","slug":"fleet-control-using-coregionalized-gaussian","title":"Fleet Control using Coregionalized Gaussian Process Policy Iteration","date":"2019-11-22","arxiv_id":"1911.10121","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-text-ranking-with-bayesian","slug":"interactive-text-ranking-with-bayesian","title":"Interactive Text Ranking with Bayesian Optimisation: A Case Study on Community QA and Summarisation","date":"2019-11-22","arxiv_id":"1911.10183","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-data-usage-via-differentiable-1","slug":"optimizing-data-usage-via-differentiable-1","title":"Optimizing Data Usage via Differentiable Rewards","date":"2019-11-22","arxiv_id":"1911.10088","repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-episodic-control","slug":"memory-efficient-episodic-control","title":"Memory-Efficient Episodic Control Reinforcement Learning with Dynamic Online k-means","date":"2019-11-21","arxiv_id":"1911.09560","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with-2","slug":"sample-efficient-reinforcement-learning-with-2","title":"Sample-Efficient Reinforcement Learning with Maximum Entropy Mellowmax Episodic Control","date":"2019-11-21","arxiv_id":"1911.09615","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-curiosity-for-efficient-exploration","slug":"bayesian-curiosity-for-efficient-exploration","title":"Bayesian Curiosity for Efficient Exploration in Reinforcement Learning","date":"2019-11-20","arxiv_id":"1911.08701","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-explicitly","slug":"deep-reinforcement-learning-with-explicitly","title":"Classification with Costly Features in Hierarchical Deep Sets","date":"2019-11-20","arxiv_id":"1911.08756","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-task-agnostic-exploration-for","slug":"evaluating-task-agnostic-exploration-for","title":"Evaluating task-agnostic exploration for fixed-batch learning of arbitrary future tasks","date":"2019-11-20","arxiv_id":"1911.08666","repositories_listed":1,"syntology":null},{"url":"/paper/neural-approximate-dynamic-programming-for-on","slug":"neural-approximate-dynamic-programming-for-on","title":"Neural Approximate Dynamic Programming for On-Demand Ride-Pooling","date":"2019-11-20","arxiv_id":"1911.08842","repositories_listed":1,"syntology":null},{"url":"/paper/cup-cluster-pruning-for-compressing-deep","slug":"cup-cluster-pruning-for-compressing-deep","title":"CUP: Cluster Pruning for Compressing Deep Neural Networks","date":"2019-11-19","arxiv_id":"1911.08630","repositories_listed":1,"syntology":null},{"url":"/paper/deep-tile-coder-an-efficient-sparse","slug":"deep-tile-coder-an-efficient-sparse","title":"Fuzzy Tiling Activations: A Simple Approach to Learning Sparse Representations Online","date":"2019-11-19","arxiv_id":"1911.08068","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-resource-allocation-in-stream","slug":"generalizable-resource-allocation-in-stream","title":"Generalizable Resource Allocation in Stream Processing via Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08517","repositories_listed":1,"syntology":null},{"url":"/paper/online-learned-continual-compression-with-1","slug":"online-learned-continual-compression-with-1","title":"Online Learned Continual Compression with Adaptive Quantization Modules","date":"2019-11-19","arxiv_id":"1911.08019","repositories_listed":1,"syntology":null},{"url":"/paper/planning-with-goal-conditioned-policies-1","slug":"planning-with-goal-conditioned-policies-1","title":"Planning with Goal-Conditioned Policies","date":"2019-11-19","arxiv_id":"1911.08453","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planning-with-goal-conditioned-policies-1#ran","syntology_url":"https://syntology.ai/paper/1911.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08453"}},"official":null}},{"url":"/paper/influence-aware-memory-for-deep-reinforcement-1","slug":"influence-aware-memory-for-deep-reinforcement-1","title":"Influence-aware Memory Architectures for Deep Reinforcement Learning","date":"2019-11-18","arxiv_id":"1911.07643","repositories_listed":1,"syntology":null},{"url":"/paper/ikea-furniture-assembly-environment-for-long","slug":"ikea-furniture-assembly-environment-for-long","title":"IKEA Furniture Assembly Environment for Long-Horizon Complex Manipulation Tasks","date":"2019-11-17","arxiv_id":"1911.07246","repositories_listed":1,"syntology":null},{"url":"/paper/missingness-as-stability-understanding-the","slug":"missingness-as-stability-understanding-the","title":"Missingness as Stability: Understanding the Structure of Missingness in Longitudinal EHR data and its Impact on Reinforcement Learning in Healthcare","date":"2019-11-16","arxiv_id":"1911.07084","repositories_listed":1,"syntology":null},{"url":"/paper/generating-persona-consistent-dialogues-by","slug":"generating-persona-consistent-dialogues-by","title":"Generating Persona Consistent Dialogues by Exploiting Natural Language Inference","date":"2019-11-14","arxiv_id":"1911.05889","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-market-making-in-a","slug":"reinforcement-learning-for-market-making-in-a","title":"Reinforcement Learning for Market Making in a Multi-agent Dealer Market","date":"2019-11-14","arxiv_id":"1911.05892","repositories_listed":1,"syntology":null},{"url":"/paper/a-convergent-off-policy-temporal-difference","slug":"a-convergent-off-policy-temporal-difference","title":"A Convergent Off-Policy Temporal Difference Algorithm","date":"2019-11-13","arxiv_id":"1911.05697","repositories_listed":1,"syntology":null},{"url":"/paper/drills-deep-reinforcement-learning-for-logic","slug":"drills-deep-reinforcement-learning-for-logic","title":"DRiLLS: Deep Reinforcement Learning for Logic Synthesis","date":"2019-11-11","arxiv_id":"1911.04021","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drills-deep-reinforcement-learning-for-logic#ran","syntology_url":"https://syntology.ai/paper/1911.04021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04021"}},"official":{"repos":["scale-lab/DRiLLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/driving-reinforcement-learning-with-models","slug":"driving-reinforcement-learning-with-models","title":"Driving Reinforcement Learning with Models","date":"2019-11-11","arxiv_id":"1911.04400","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-connected-autonomous-driving","slug":"multi-agent-connected-autonomous-driving","title":"Multi-Agent Connected Autonomous Driving using Deep Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04175","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-connected-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/1911.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04175"}},"official":{"repos":["praveen-palanisamy/macad-gym"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smix-enhancing-centralized-value-functions","slug":"smix-enhancing-centralized-value-functions","title":"SMIX($λ$): Enhancing Centralized Value Functions for Cooperative Multi-Agent Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04094","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforced-generation-of-adversarial","slug":"a-reinforced-generation-of-adversarial","title":"A Reinforced Generation of Adversarial Examples for Neural Machine Translation","date":"2019-11-09","arxiv_id":"1911.03677","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-common-question-generation-from","slug":"unsupervised-common-question-generation-from","title":"Contrastive Multi-document Question Generation","date":"2019-11-08","arxiv_id":"1911.03047","repositories_listed":1,"syntology":null},{"url":"/paper/h_inf-model-free-reinforcement-learning-with","slug":"h_inf-model-free-reinforcement-learning-with","title":"$H_\\infty$ Model-free Reinforcement Learning with Robust Stability Guarantee","date":"2019-11-07","arxiv_id":"1911.02875","repositories_listed":1,"syntology":null},{"url":"/paper/improving-reinforcement-learning-algorithms","slug":"improving-reinforcement-learning-algorithms","title":"Improving reinforcement learning algorithms: towards optimal learning rate policies","date":"2019-11-06","arxiv_id":"1911.02319","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-based-approach","slug":"a-deep-reinforcement-learning-based-approach","title":"A Deep Reinforcement Learning Approach to First-Order Logic Theorem Proving","date":"2019-11-05","arxiv_id":"1911.02065","repositories_listed":1,"syntology":null},{"url":"/paper/gym-ignition-reproducible-robotic-simulations","slug":"gym-ignition-reproducible-robotic-simulations","title":"Gym-Ignition: Reproducible Robotic Simulations for Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.01715","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-numeric-concepts-in-multi-agent","slug":"emergence-of-numeric-concepts-in-multi-agent","title":"Emergence of Numeric Concepts in Multi-Agent Autonomous Communication","date":"2019-11-04","arxiv_id":"1911.01098","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-trajectories-via-subgoal-1","slug":"learning-from-trajectories-via-subgoal-1","title":"Learning from Trajectories via Subgoal Discovery","date":"2019-11-03","arxiv_id":"1911.07224","repositories_listed":1,"syntology":null},{"url":"/paper/on-solving-the-2-dimensional-greedy-shooter","slug":"on-solving-the-2-dimensional-greedy-shooter","title":"On Solving the 2-Dimensional Greedy Shooter Problem for UAVs","date":"2019-11-02","arxiv_id":"1911.01419","repositories_listed":1,"syntology":null},{"url":"/paper/thompson-sampling-for-contextual-bandit","slug":"thompson-sampling-for-contextual-bandit","title":"Thompson Sampling for Contextual Bandit Problems with Auxiliary Safety Constraints","date":"2019-11-02","arxiv_id":"1911.00638","repositories_listed":1,"syntology":null},{"url":"/paper/explicit-explore-exploit-algorithms-in","slug":"explicit-explore-exploit-algorithms-in","title":"Explicit Explore-Exploit Algorithms in Continuous State Spaces","date":"2019-11-01","arxiv_id":"1911.00617","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-speedy-q-learning","slug":"generalized-speedy-q-learning","title":"Generalized Speedy Q-learning","date":"2019-11-01","arxiv_id":"1911.00397","repositories_listed":1,"syntology":null},{"url":"/paper/positive-unlabeled-reward-learning","slug":"positive-unlabeled-reward-learning","title":"Positive-Unlabeled Reward Learning","date":"2019-11-01","arxiv_id":"1911.00459","repositories_listed":1,"syntology":null},{"url":"/paper/cascaded-lstms-based-deep-reinforcement","slug":"cascaded-lstms-based-deep-reinforcement","title":"Cascaded LSTMs based Deep Reinforcement Learning for Goal-driven Dialogue","date":"2019-10-31","arxiv_id":"1910.14229","repositories_listed":1,"syntology":null},{"url":"/paper/continual-unsupervised-representation","slug":"continual-unsupervised-representation","title":"Continual Unsupervised Representation Learning","date":"2019-10-31","arxiv_id":"1910.14481","repositories_listed":1,"syntology":null},{"url":"/paper/nat-neural-architecture-transformer-for","slug":"nat-neural-architecture-transformer-for","title":"NAT: Neural Architecture Transformer for Accurate and Compact Architectures","date":"2019-10-31","arxiv_id":"1910.14488","repositories_listed":1,"syntology":null},{"url":"/paper/policy-continuation-with-hindsight-inverse","slug":"policy-continuation-with-hindsight-inverse","title":"Policy Continuation with Hindsight Inverse Dynamics","date":"2019-10-30","arxiv_id":"1910.14055","repositories_listed":1,"syntology":null},{"url":"/paper/191013249","slug":"191013249","title":"Navigation Agents for the Visually Impaired: A Sidewalk Simulator and Experiments","date":"2019-10-29","arxiv_id":"1910.13249","repositories_listed":1,"syntology":null},{"url":"/paper/191013406","slug":"191013406","title":"Generalization of Reinforcement Learners with Working and Episodic Memory","date":"2019-10-29","arxiv_id":"1910.13406","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-deep-energy-based","slug":"a-framework-for-deep-energy-based","title":"Quantum enhancements for deep reinforcement learning in large spaces","date":"2019-10-28","arxiv_id":"1910.12760","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-methods-for-model-based","slug":"asynchronous-methods-for-model-based","title":"Asynchronous Methods for Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12453","repositories_listed":1,"syntology":null},{"url":"/paper/entity-abstraction-in-visual-model-based","slug":"entity-abstraction-in-visual-model-based","title":"Entity Abstraction in Visual Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/entity-abstraction-in-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/1910.12827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12827"}},"official":{"repos":["jcoreyes/OP3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-in-reinforcement-learning-with","slug":"generalization-in-reinforcement-learning-with","title":"Generalization in Reinforcement Learning with Selective Noise Injection and Information Bottleneck","date":"2019-10-28","arxiv_id":"1910.12911","repositories_listed":1,"syntology":null},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/long-term-joint-scheduling-for-urban-traffic","slug":"long-term-joint-scheduling-for-urban-traffic","title":"Long-term Joint Scheduling for Urban Traffic","date":"2019-10-27","arxiv_id":"1910.12283","repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","repositories_listed":1,"syntology":null},{"url":"/paper/convergent-policy-optimization-for-safe","slug":"convergent-policy-optimization-for-safe","title":"Convergent Policy Optimization for Safe Reinforcement Learning","date":"2019-10-26","arxiv_id":"1910.12156","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-hol4","slug":"deep-reinforcement-learning-in-hol4","title":"Deep Reinforcement Learning for Synthesizing Functions in Higher-Order Logic","date":"2019-10-25","arxiv_id":"1910.11797","repositories_listed":1,"syntology":null},{"url":"/paper/relay-policy-learning-solving-long-horizon","slug":"relay-policy-learning-solving-long-horizon","title":"Relay Policy Learning: Solving Long-Horizon Tasks via Imitation and Reinforcement Learning","date":"2019-10-25","arxiv_id":"1910.11956","repositories_listed":1,"syntology":null},{"url":"/paper/hrl4in-hierarchical-reinforcement-learning","slug":"hrl4in-hierarchical-reinforcement-learning","title":"HRL4IN: Hierarchical Reinforcement Learning for Interactive Navigation with Mobile Manipulators","date":"2019-10-24","arxiv_id":"1910.11432","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hrl4in-hierarchical-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.11432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11432"}},"official":null}},{"url":"/paper/learning-hierarchical-control-for-robust-in","slug":"learning-hierarchical-control-for-robust-in","title":"Learning Hierarchical Control for Robust In-Hand Manipulation","date":"2019-10-24","arxiv_id":"1910.10985","repositories_listed":1,"syntology":null},{"url":"/paper/pun-gan-generative-adversarial-network-for","slug":"pun-gan-generative-adversarial-network-for","title":"Pun-GAN: Generative Adversarial Network for Pun Generation","date":"2019-10-24","arxiv_id":"1910.10950","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-curiosity-driven-exploration","slug":"attention-based-curiosity-driven-exploration","title":"Attention-based Curiosity-driven Exploration in Deep Reinforcement Learning","date":"2019-10-23","arxiv_id":"1910.10840","repositories_listed":1,"syntology":null},{"url":"/paper/autoencoding-with-xcsf","slug":"autoencoding-with-xcsf","title":"Autoencoding with a Classifier System","date":"2019-10-23","arxiv_id":"1910.10579","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-imagined-goals-for-self-supervised","slug":"contextual-imagined-goals-for-self-supervised","title":"Contextual Imagined Goals for Self-Supervised Robotic Learning","date":"2019-10-23","arxiv_id":"1910.11670","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-imagined-goals-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/1910.11670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11670"}},"official":null}},{"url":"/paper/efficient-decoupled-neural-architecture","slug":"efficient-decoupled-neural-architecture","title":"Efficient Decoupled Neural Architecture Search by Structure and Operation Sampling","date":"2019-10-23","arxiv_id":"1910.10397","repositories_listed":1,"syntology":null},{"url":"/paper/bottom-up-meta-policy-search","slug":"bottom-up-meta-policy-search","title":"Bottom-Up Meta-Policy Search","date":"2019-10-22","arxiv_id":"1910.10232","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-graph-walk-for-semi-supervised","slug":"collaborative-graph-walk-for-semi-supervised","title":"Collaborative Graph Walk for Semi-supervised Multi-Label Node Classification","date":"2019-10-22","arxiv_id":"1910.09706","repositories_listed":1,"syntology":null},{"url":"/paper/improving-the-gating-mechanism-of-recurrent-1","slug":"improving-the-gating-mechanism-of-recurrent-1","title":"Improving the Gating Mechanism of Recurrent Neural Networks","date":"2019-10-22","arxiv_id":"1910.09890","repositories_listed":1,"syntology":null},{"url":"/paper/learning-humanoid-robot-running-skills","slug":"learning-humanoid-robot-running-skills","title":"Learning Humanoid Robot Running Skills through Proximal Policy Optimization","date":"2019-10-22","arxiv_id":"1910.10620","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-attention-walk-for-semi-supervised","slug":"recurrent-attention-walk-for-semi-supervised","title":"Recurrent Attention Walk for Semi-supervised Classification","date":"2019-10-22","arxiv_id":"1910.10266","repositories_listed":1,"syntology":null},{"url":"/paper/teach-biped-robots-to-walk-via-gait","slug":"teach-biped-robots-to-walk-via-gait","title":"Teach Biped Robots to Walk via Gait Principles and Reinforcement Learning with Adversarial Critics","date":"2019-10-22","arxiv_id":"1910.10194","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-skill-networks-unsupervised-robot","slug":"adversarial-skill-networks-unsupervised-robot","title":"Adversarial Skill Networks: Unsupervised Robot Skill Learning from Video","date":"2019-10-21","arxiv_id":"1910.09430","repositories_listed":1,"syntology":null},{"url":"/paper/dealing-with-sparse-rewards-in-reinforcement","slug":"dealing-with-sparse-rewards-in-reinforcement","title":"Dealing with Sparse Rewards in Reinforcement Learning","date":"2019-10-21","arxiv_id":"1910.09281","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-control-of","slug":"deep-reinforcement-learning-control-of","title":"Deep Reinforcement Learning Control of Quantum Cartpoles","date":"2019-10-21","arxiv_id":"1910.09200","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-via-sample-efficient-subgoal","slug":"exploration-via-sample-efficient-subgoal","title":"Dynamic Subgoal-based Exploration via Bayesian Optimization","date":"2019-10-21","arxiv_id":"1910.09143","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-map-natural-language-instructions","slug":"learning-to-map-natural-language-instructions","title":"Learning to Map Natural Language Instructions to Physical Quadcopter Control using Simulated Flight","date":"2019-10-21","arxiv_id":"1910.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-map-natural-language-instructions#ran","syntology_url":"https://syntology.ai/paper/1910.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09664"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rlscheduler-learn-to-schedule-hpc-batch-jobs","slug":"rlscheduler-learn-to-schedule-hpc-batch-jobs","title":"RLScheduler: An Automated HPC Batch Job Scheduler Using Reinforcement Learning","date":"2019-10-20","arxiv_id":"1910.08925","repositories_listed":1,"syntology":null},{"url":"/paper/a-structured-prediction-approach-for","slug":"a-structured-prediction-approach-for","title":"A Structured Prediction Approach for Generalization in Cooperative Multi-Agent Reinforcement Learning","date":"2019-10-19","arxiv_id":"1910.08809","repositories_listed":1,"syntology":null},{"url":"/paper/natural-question-generation-with","slug":"natural-question-generation-with","title":"Natural Question Generation with Reinforcement Learning Based Graph-to-Sequence Model","date":"2019-10-19","arxiv_id":"1910.08832","repositories_listed":1,"syntology":null},{"url":"/paper/towards-more-sample-efficiency","slug":"towards-more-sample-efficiency","title":"Towards More Sample Efficiency in Reinforcement Learning with Data Augmentation","date":"2019-10-19","arxiv_id":"1910.09959","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-data-augmentation-by-learning-the","slug":"automatic-data-augmentation-by-learning-the","title":"Automatic Data Augmentation by Learning the Deterministic Policy","date":"2019-10-18","arxiv_id":"1910.08343","repositories_listed":1,"syntology":null},{"url":"/paper/first-order-preconditioning-via-hypergradient","slug":"first-order-preconditioning-via-hypergradient","title":"First-Order Preconditioning via Hypergradient Descent","date":"2019-10-18","arxiv_id":"1910.08461","repositories_listed":1,"syntology":null},{"url":"/paper/multi-view-reinforcement-learning","slug":"multi-view-reinforcement-learning","title":"Multi-View Reinforcement Learning","date":"2019-10-18","arxiv_id":"1910.08285","repositories_listed":1,"syntology":null},{"url":"/paper/rtfm-generalising-to-novel-environment","slug":"rtfm-generalising-to-novel-environment","title":"RTFM: Generalising to Novel Environment Dynamics via Reading","date":"2019-10-18","arxiv_id":"1910.08210","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rtfm-generalising-to-novel-environment#ran","syntology_url":"https://syntology.ai/paper/1910.08210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.08210"}},"official":null}},{"url":"/paper/adaptive-curriculum-generation-from","slug":"adaptive-curriculum-generation-from","title":"Adaptive Curriculum Generation from Demonstrations for Sim-to-Real Visuomotor Control","date":"2019-10-17","arxiv_id":"1910.07972","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-discretization-for-episodic","slug":"adaptive-discretization-for-episodic","title":"Adaptive Discretization for Episodic Reinforcement Learning in Metric Spaces","date":"2019-10-17","arxiv_id":"1910.08151","repositories_listed":1,"syntology":null}],"record_sha256":"ad0c9fce689507bc89d45133e3870dec70c78e8f0ca0414fe43d1b903608ee2e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}