{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/116","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":116,"pages_in_order":135,"rows_per_page":100,"rows":[11501,11600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/115","next":"/task/reinforcement-learning-2/papers/117","papers":[{"url":null,"slug":"fair-loss-margin-aware-reinforcement-learning","title":"Fair Loss: Margin-Aware Reinforcement Learning for Deep Face Recognition","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-in-generation-a-closer-look-at","title":"Generalization in Generation: A closer look at Exposure Bias","date":"2019-10-01","arxiv_id":"1910.00292","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-paraphrases-with-lean-vocabulary","title":"Generating Paraphrases with Lean Vocabulary","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-translation-for-machines-the","title":"Machine Translation for Machines: the Sentiment Classification Use Case","date":"2019-10-01","arxiv_id":"1910.00478","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-multi-objective","title":"Reinforcement Learning for Multi-Objective Optimization of Online Decisions in High-Dimensional Systems","date":"2019-10-01","arxiv_id":"1910.00211","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-interaction-aware-scene-understanding","title":"Dynamic Interaction-Aware Scene Understanding for Reinforcement Learning in Autonomous Driving","date":"2019-09-30","arxiv_id":"1909.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-meta-reinforcement-learning-via-1","title":"MGHRL: Meta Goal-generation for Hierarchical Reinforcement Learning","date":"2019-09-30","arxiv_id":"1909.13607","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-motion-planning-of-quadrotors","title":"End-to-End Motion Planning of Quadrotors Using Deep Reinforcement Learning","date":"2019-09-30","arxiv_id":"1909.13599","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlcache-automated-cache-management-using","title":"RLCache: Automated Cache Management Using Reinforcement Learning","date":"2019-09-30","arxiv_id":"1909.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"tensor-based-cooperative-control-for-large","title":"Tensor-based Cooperative Control for Large Scale Multi-intersection Traffic Signal Using Deep Reinforcement Learning and Imitation Learning","date":"2019-09-30","arxiv_id":"1909.13428","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-the-computation-of-ucb-and","title":"Accelerating the Computation of UCB and Related Indices for Reinforcement Learning","date":"2019-09-28","arxiv_id":"1909.13158","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-power","title":"Deep Reinforcement Learning Based Power control for Wireless Multicast Systems","date":"2019-09-27","arxiv_id":"1910.05308","repositories_listed":0,"syntology":null},{"url":null,"slug":"interaction-aware-multi-agent-reinforcement","title":"Interaction-Aware Multi-Agent Reinforcement Learning for Mobile Agents with Individual Goals","date":"2019-09-27","arxiv_id":"1909.12925","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-atari-ball-games-with-hierarchical","title":"Playing Atari Ball Games with Hierarchical Reinforcement Learning","date":"2019-09-27","arxiv_id":"1909.12465","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-on-autonomous","title":"Safe Reinforcement Learning on Autonomous Vehicles","date":"2019-09-27","arxiv_id":"1910.00399","repositories_listed":0,"syntology":null},{"url":null,"slug":"surreal-system-fully-integrated-stack-for","title":"SURREAL-System: Fully-Integrated Stack for Distributed Deep Reinforcement Learning","date":"2019-09-27","arxiv_id":"1909.12989","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-dimensional-control-using-generalized","title":"MERL: Multi-Head Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.11939","repositories_listed":0,"syntology":null},{"url":null,"slug":"relationship-explainable-multi-objective","title":"Relationship Explainable Multi-objective Reinforcement Learning with Semantic Explainability Generation","date":"2019-09-26","arxiv_id":"1909.12268","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapt-to-learn-policy-transfer-in","title":"Adapt-to-Learn: Policy Transfer in Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"advantage-weighted-regression-simple-and-1","title":"Advantage Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-generalization-in-td-methods-for","title":"Assessing Generalization in TD methods for Deep Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-privileged-reinforcement-learning","title":"Attention Privileged Reinforcement Learning for Domain Transfer","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-policy-gradient-methods-for","title":"AUGMENTED POLICY GRADIENT METHODS FOR EFFICIENT REINFORCEMENT LEARNING","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bananas-bayesian-optimization-with-neural-1","title":"BANANAS: Bayesian Optimization with Neural Networks for Neural Architecture Search","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-guided-reinforcement-learning","title":"Behavior-Guided Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"capacity-limited-reinforcement-learning","title":"CAPACITY-LIMITED REINFORCEMENT LEARNING: APPLICATIONS IN DEEP ACTOR-CRITIC METHODS FOR CONTINUOUS CONTROL","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"city-metro-network-expansion-with","title":"City Metro Network Expansion with Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-inter-agent-knowledge","title":"Collaborative Inter-agent Knowledge Distillation for Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"consistent-meta-reinforcement-learning-via","title":"Consistent Meta-Reinforcement Learning via Model Identification and Experience Relabeling","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-inverse-reinforcement-learning","title":"Contextual Inverse Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-regularization-for-model-based","title":"Counterfactual Regularization for Model-Based Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"crossnorm-on-normalization-for-off-policy","title":"CrossNorm: On Normalization for Off-Policy Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deepagrel-biologically-plausible-deep","title":"DeepAGREL: Biologically plausible deep learning via direct reinforcement","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"do-recent-advancements-in-model-based-deep","title":"Do recent advancements in model-based deep reinforcement learning really improve data efficiency?","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-meta-reinforcement-learning-via","title":"Efficient meta reinforcement learning via meta goal generation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"event-discovery-for-history-representation-in","title":"Event Discovery for History Representation in Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-reinforcement-learning-to-unseen","title":"Generalizing Reinforcement Learning to Unseen Actions","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hippocampal-neuronal-representations-in","title":"HIPPOCAMPAL NEURONAL REPRESENTATIONS IN CONTINUAL LEARNING","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hope-for-the-best-but-prepare-for-the-worst","title":"Hope For The Best But Prepare For The Worst: Cautious Adaptation In RL Agents","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-many-weights-are-enough-can-tensor","title":"How many weights are enough : can tensor factorization learn efficient policies ?","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-exploration-of-deep-reinforcement","title":"Improving Exploration of Deep Reinforcement Learning using Planning for Policy Search","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-sat-solver-heuristics-with-graph-1","title":"Improving SAT Solver Heuristics with Graph Networks and Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-algorithmic-solutions-to-symbolic-1","title":"Learning Algorithmic Solutions to Symbolic Planning Tasks with a Neural Computer","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-functionally-decomposed-hierarchies","title":"Learning Functionally Decomposed Hierarchies for Continuous Navigation Tasks","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-good-policies-by-learning-good","title":"Learning Good Policies By Learning Good Perceptual Models","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-key-steps-to-attack-deep","title":"Learning Key Steps to Attack Deep Reinforcement Learning Agents","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-temporal-abstraction-with","title":"Learning Temporal Abstraction with Information-theoretic Constraints for Hierarchical Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reach-goals-without-reinforcement","title":"Learning to Reach Goals Without Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason-distilling-hierarchy-via","title":"Learning to Reason: Distilling Hierarchy via Self-Supervision and Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-world-graph-decompositions-to","title":"Learning World Graph Decompositions To Accelerate Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-via-learned-loss-1","title":"Meta Learning via Learned Loss","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mint-matrix-interleaving-for-multi-task","title":"Mint: Matrix-Interleaving for Multi-Task Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-ensemble-based-intrinsic-reward-for","title":"Model Ensemble-Based Intrinsic Reward for Sparse Reward Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-imitation-for-model-based-reinforcement","title":"Model Imitation for Model-Based Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11821","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-fake-news-in-social-networks-with","title":"Modeling Fake News in Social Networks with Deep Multi-Agent Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"moet-interpretable-and-verifiable-1","title":"MoET: Interpretable and Verifiable Reinforcement Learning via Mixture of Expert Trees","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-hierarchical-reinforcement-1","title":"Multi-Agent Hierarchical Reinforcement Learning for Humanoid Navigation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-step-greedy-policies-in-model-free-deep","title":"Multi-step Greedy Policies in Model-Free Deep Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multiagent-reinforcement-learning-in-games","title":"Multiagent Reinforcement Learning in Games with an Iterated Dominance Solution","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-by-local-improvement","title":"Policy Optimization by Local Improvement through Search","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-tree-network","title":"Policy Tree Network","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-as-batch-meta-reinforcement","title":"Multi-task Batch Reinforcement Learning with Metric Learning","date":"2019-09-25","arxiv_id":"1909.11373","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-as-batch-meta-reinforcement-1","title":"Pre-training as Batch Meta Reinforcement Learning with tiMe","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-view-of-multi-agent","title":"Probabilistic View of Multi-agent Reinforcement Learning: A Unified Approach","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"qxplore-q-learning-exploration-by-maximizing-1","title":"QXplore: Q-Learning Exploration by Maximizing Temporal Difference Error","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-monte-carlo-tree-search-agents-by","title":"REFINING MONTE CARLO TREE SEARCH AGENTS BY MONTE CARLO TREE SEARCH","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-suppression-of","title":"Reinforcement learning for suppression of collective activity in oscillatory ensembles","date":"2019-09-25","arxiv_id":"1909.12154","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-chromatic-1","title":"Reinforcement Learning with Chromatic Networks","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-value-expansion-for-sample","title":"Risk Averse Value Expansion for Sample Efficient and Robust Policy Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-domain-randomization-for-reinforcement-1","title":"Robust Domain Randomization for Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"s2vg-soft-stochastic-value-gradient-method","title":"S2VG: Soft Stochastic Value Gradient method","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-intrinsic-exploration-model","title":"Sequence-level Intrinsic Exploration Model for Partially Observable Domains","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-single-objective-tasks-by-preference","title":"Solving single-objective tasks by preference multi-objective reinforcement learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-skill-coding-learning-behavioral","title":"Sparse Skill Coding: Learning Behavioral Hierarchies with Sparse Codes","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-off-policy-reinforcement-learning-1","title":"Stabilizing Off-Policy Reinforcement Learning with Conservative Policy Gradients","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"striving-for-simplicity-in-off-policy-deep-1","title":"Striving for Simplicity in Off-Policy Deep Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"subjective-reinforcement-learning-for-open","title":"Subjective Reinforcement Learning for Open Complex Environments","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-difference-weighted-ensemble-for","title":"Temporal Difference Weighted Ensemble For Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-simplicity-in-deep-reinforcement","title":"Towards Simplicity in Deep Reinforcement Learning: Streamlined Off-Policy Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"training-a-constrained-natural-media-painting","title":"Training a Constrained Natural Media Painting Agent using Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-constrained-reinforcement","title":"Variational Constrained Reinforcement Learning with Application to Planning at Roundabout","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-an-autonomous-vehicle-with-deep","title":"Controlling an Autonomous Vehicle with Deep Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.12153","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-inference-and-exploration-for-1","title":"Efficient Inference and Exploration for Reinforcement Learning","date":"2019-09-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"power-allocation-in-cache-aided-noma-systems","title":"Power Allocation in Cache-Aided NOMA Systems: Optimization and Deep Reinforcement Learning Approaches","date":"2019-09-24","arxiv_id":"1909.11074","repositories_listed":0,"syntology":null},{"url":null,"slug":"190910400","title":"Robot Navigation in Crowds by Graph Convolutional Networks with Attention Learned from Human Gaze","date":"2019-09-23","arxiv_id":"1909.10400","repositories_listed":0,"syntology":null},{"url":null,"slug":"190910449","title":"PAC Reinforcement Learning without Real-World Feedback","date":"2019-09-23","arxiv_id":"1909.10449","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-attractor-selection-using-deep","title":"Constrained Attractor Selection Using Deep Reinforcement Learning","date":"2019-09-23","arxiv_id":"1909.10500","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-independent-and-centralized-multi","title":"Integrating independent and centralized multi-agent reinforcement learning for traffic signal network optimization","date":"2019-09-23","arxiv_id":"1909.10651","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-does-hierarchy-sometimes-work-so-well-in","title":"Why Does Hierarchy (Sometimes) Work So Well in Reinforcement Learning?","date":"2019-09-23","arxiv_id":"1909.10618","repositories_listed":0,"syntology":null},{"url":null,"slug":"190909906","title":"Leveraging Human Guidance for Deep Reinforcement Learning Tasks","date":"2019-09-21","arxiv_id":"1909.09906","repositories_listed":0,"syntology":null},{"url":null,"slug":"190909705","title":"A Layered Architecture for Active Perception: Image Classification using Deep Reinforcement Learning","date":"2019-09-20","arxiv_id":"1909.09705","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-much-do-unstated-problem-constraints","title":"How Much Do Unstated Problem Constraints Limit Deep Robotic Reinforcement Learning?","date":"2019-09-20","arxiv_id":"1909.09282","repositories_listed":0,"syntology":null},{"url":null,"slug":"redirection-controller-using-reinforcement","title":"Redirection Controller Using Reinforcement Learning","date":"2019-09-20","arxiv_id":"1909.09505","repositories_listed":0,"syntology":null},{"url":null,"slug":"macs-deep-reinforcement-learning-based-sdn","title":"MACS: Deep Reinforcement Learning based SDN Controller Synchronization Policy Design","date":"2019-09-19","arxiv_id":"1909.09063","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-sound-interpretation-combining-sight","title":"Robot Sound Interpretation: Combining Sight and Sound in Learning-Based Control","date":"2019-09-19","arxiv_id":"1909.09172","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-function-estimation-in-markov-reward","title":"Instance-dependent $\\ell_\\infty$-bounds for policy evaluation in tabular reinforcement learning","date":"2019-09-19","arxiv_id":"1909.08749","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-two-tier-approach-to-hyper","title":"A Hierarchical Two-tier Approach to Hyper-parameter Optimization in Reinforcement Learning","date":"2019-09-18","arxiv_id":"1909.08332","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-lane-change-decision-making-using","title":"Automated Lane Change Decision Making using Deep Reinforcement Learning in Dynamic and Uncertain Highway Environment","date":"2019-09-18","arxiv_id":"1909.11538","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepgait-planning-and-control-of-quadrupedal","title":"DeepGait: Planning and Control of Quadrupedal Gaits using Deep Reinforcement Learning","date":"2019-09-18","arxiv_id":"1909.08399","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-aware-computation-offloading-in","title":"Dependency-Aware Computation Offloading in Mobile Edge Computing: A Reinforcement Learning Approach","date":"2019-09-18","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"e52ad6f583de67502d367d24730e5a00b63d1de2c6d3700852285dfb292155c2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}