{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/130","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":130,"pages_in_order":152,"rows_per_page":100,"rows":[12901,13000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/129","next":"/task/reinforcement-learning-1/papers/131","papers":[{"url":null,"slug":"city-metro-network-expansion-with","title":"City Metro Network Expansion with Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-inter-agent-knowledge","title":"Collaborative Inter-agent Knowledge Distillation for Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"consistent-meta-reinforcement-learning-via","title":"Consistent Meta-Reinforcement Learning via Model Identification and Experience Relabeling","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-inverse-reinforcement-learning","title":"Contextual Inverse Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-regularization-for-model-based","title":"Counterfactual Regularization for Model-Based Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"crossnorm-on-normalization-for-off-policy","title":"CrossNorm: On Normalization for Off-Policy Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-rl-for-blood-glucose-control-lessons","title":"Deep RL for Blood Glucose Control: Lessons, Challenges, and Opportunities","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deepagrel-biologically-plausible-deep","title":"DeepAGREL: Biologically plausible deep learning via direct reinforcement","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"do-recent-advancements-in-model-based-deep","title":"Do recent advancements in model-based deep reinforcement learning really improve data efficiency?","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-meta-reinforcement-learning-via","title":"Efficient meta reinforcement learning via meta goal generation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"event-discovery-for-history-representation-in","title":"Event Discovery for History Representation in Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evo-nas-evolutionary-neural-hybrid-agent-for","title":"Evo-NAS: Evolutionary-Neural Hybrid Agent for Architecture Search","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-reinforcement-learning-to-unseen","title":"Generalizing Reinforcement Learning to Unseen Actions","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hippocampal-neuronal-representations-in","title":"HIPPOCAMPAL NEURONAL REPRESENTATIONS IN CONTINUAL LEARNING","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hope-for-the-best-but-prepare-for-the-worst","title":"Hope For The Best But Prepare For The Worst: Cautious Adaptation In RL Agents","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-many-weights-are-enough-can-tensor","title":"How many weights are enough : can tensor factorization learn efficient policies ?","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-exploration-of-deep-reinforcement","title":"Improving Exploration of Deep Reinforcement Learning using Planning for Policy Search","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-sat-solver-heuristics-with-graph-1","title":"Improving SAT Solver Heuristics with Graph Networks and Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-algorithmic-solutions-to-symbolic-1","title":"Learning Algorithmic Solutions to Symbolic Planning Tasks with a Neural Computer","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-by-shaking-computing-policy","title":"Learning by shaking: Computing policy gradients by physical forward-propagation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-functionally-decomposed-hierarchies","title":"Learning Functionally Decomposed Hierarchies for Continuous Navigation Tasks","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-good-policies-by-learning-good","title":"Learning Good Policies By Learning Good Perceptual Models","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-key-steps-to-attack-deep","title":"Learning Key Steps to Attack Deep Reinforcement Learning Agents","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-temporal-abstraction-with","title":"Learning Temporal Abstraction with Information-theoretic Constraints for Hierarchical Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reach-goals-without-reinforcement","title":"Learning to Reach Goals Without Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason-distilling-hierarchy-via","title":"Learning to Reason: Distilling Hierarchy via Self-Supervision and Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-social-influence-through","title":"Learning with Social Influence through Interior Policy Differentiation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-world-graph-decompositions-to","title":"Learning World Graph Decompositions To Accelerate Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-planning-short-term-adjustments","title":"Long-term planning, short-term adjustments","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-via-learned-loss-1","title":"Meta Learning via Learned Loss","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mint-matrix-interleaving-for-multi-task","title":"Mint: Matrix-Interleaving for Multi-Task Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-ensemble-based-intrinsic-reward-for","title":"Model Ensemble-Based Intrinsic Reward for Sparse Reward Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-learning-control-of-nonlinear","title":"Model-free Learning Control of Nonlinear Stochastic Systems with Stability Guarantee","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-imitation-for-model-based-reinforcement","title":"Model Imitation for Model-Based Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11821","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-fake-news-in-social-networks-with","title":"Modeling Fake News in Social Networks with Deep Multi-Agent Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"moet-interpretable-and-verifiable-1","title":"MoET: Interpretable and Verifiable Reinforcement Learning via Mixture of Expert Trees","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-hierarchical-reinforcement-1","title":"Multi-Agent Hierarchical Reinforcement Learning for Humanoid Navigation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-step-greedy-policies-in-model-free-deep","title":"Multi-step Greedy Policies in Model-Free Deep Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multiagent-reinforcement-learning-in-games","title":"Multiagent Reinforcement Learning in Games with an Iterated Dominance Solution","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-simulation-for-imitation-learning","title":"Partial Simulation for Imitation Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-by-local-improvement","title":"Policy Optimization by Local Improvement through Search","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-tree-network","title":"Policy Tree Network","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-as-batch-meta-reinforcement","title":"Multi-task Batch Reinforcement Learning with Metric Learning","date":"2019-09-25","arxiv_id":"1909.11373","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-as-batch-meta-reinforcement-1","title":"Pre-training as Batch Meta Reinforcement Learning with tiMe","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-view-of-multi-agent","title":"Probabilistic View of Multi-agent Reinforcement Learning: A Unified Approach","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"qxplore-q-learning-exploration-by-maximizing-1","title":"QXplore: Q-Learning Exploration by Maximizing Temporal Difference Error","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-monte-carlo-tree-search-agents-by","title":"REFINING MONTE CARLO TREE SEARCH AGENTS BY MONTE CARLO TREE SEARCH","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-suppression-of","title":"Reinforcement learning for suppression of collective activity in oscillatory ensembles","date":"2019-09-25","arxiv_id":"1909.12154","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-chromatic-1","title":"Reinforcement Learning with Chromatic Networks","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-value-expansion-for-sample","title":"Risk Averse Value Expansion for Sample Efficient and Robust Policy Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-domain-randomization-for-reinforcement-1","title":"Robust Domain Randomization for Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"s2vg-soft-stochastic-value-gradient-method","title":"S2VG: Soft Stochastic Value Gradient method","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-intrinsic-exploration-model","title":"Sequence-level Intrinsic Exploration Model for Partially Observable Domains","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-single-objective-tasks-by-preference","title":"Solving single-objective tasks by preference multi-objective reinforcement learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-skill-coding-learning-behavioral","title":"Sparse Skill Coding: Learning Behavioral Hierarchies with Sparse Codes","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-off-policy-reinforcement-learning-1","title":"Stabilizing Off-Policy Reinforcement Learning with Conservative Policy Gradients","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"striving-for-simplicity-in-off-policy-deep-1","title":"Striving for Simplicity in Off-Policy Deep Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"subjective-reinforcement-learning-for-open","title":"Subjective Reinforcement Learning for Open Complex Environments","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-difference-weighted-ensemble-for","title":"Temporal Difference Weighted Ensemble For Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-simplicity-in-deep-reinforcement","title":"Towards Simplicity in Deep Reinforcement Learning: Streamlined Off-Policy Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"training-a-constrained-natural-media-painting","title":"Training a Constrained Natural Media Painting Agent using Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-representation-learning-for-multi","title":"Trajectory representation learning for Multi-Task NMRDPs planning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-constrained-reinforcement","title":"Variational Constrained Reinforcement Learning with Application to Planning at Roundabout","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-policy-transfer-with-disentangled","title":"Zero-Shot Policy Transfer with Disentangled Attention","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"brain-inspired-hardware-for-artificial","title":"Brain-Inspired Hardware for Artificial Intelligence: Accelerated Learning in a Physical-Model Spiking Neural Network","date":"2019-09-24","arxiv_id":"1909.11145","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-an-autonomous-vehicle-with-deep","title":"Controlling an Autonomous Vehicle with Deep Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.12153","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-inference-and-exploration-for-1","title":"Efficient Inference and Exploration for Reinforcement Learning","date":"2019-09-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"power-allocation-in-cache-aided-noma-systems","title":"Power Allocation in Cache-Aided NOMA Systems: Optimization and Deep Reinforcement Learning Approaches","date":"2019-09-24","arxiv_id":"1909.11074","repositories_listed":0,"syntology":null},{"url":null,"slug":"190910304","title":"Where to Look Next: Unsupervised Active Visual Exploration on 360° Input","date":"2019-09-23","arxiv_id":"1909.10304","repositories_listed":0,"syntology":null},{"url":null,"slug":"190910400","title":"Robot Navigation in Crowds by Graph Convolutional Networks with Attention Learned from Human Gaze","date":"2019-09-23","arxiv_id":"1909.10400","repositories_listed":0,"syntology":null},{"url":null,"slug":"190910449","title":"PAC Reinforcement Learning without Real-World Feedback","date":"2019-09-23","arxiv_id":"1909.10449","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-attractor-selection-using-deep","title":"Constrained Attractor Selection Using Deep Reinforcement Learning","date":"2019-09-23","arxiv_id":"1909.10500","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-independent-and-centralized-multi","title":"Integrating independent and centralized multi-agent reinforcement learning for traffic signal network optimization","date":"2019-09-23","arxiv_id":"1909.10651","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-does-hierarchy-sometimes-work-so-well-in","title":"Why Does Hierarchy (Sometimes) Work So Well in Reinforcement Learning?","date":"2019-09-23","arxiv_id":"1909.10618","repositories_listed":0,"syntology":null},{"url":null,"slug":"190909906","title":"Leveraging Human Guidance for Deep Reinforcement Learning Tasks","date":"2019-09-21","arxiv_id":"1909.09906","repositories_listed":0,"syntology":null},{"url":null,"slug":"190909705","title":"A Layered Architecture for Active Perception: Image Classification using Deep Reinforcement Learning","date":"2019-09-20","arxiv_id":"1909.09705","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-much-do-unstated-problem-constraints","title":"How Much Do Unstated Problem Constraints Limit Deep Robotic Reinforcement Learning?","date":"2019-09-20","arxiv_id":"1909.09282","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-of-approximate-and","title":"On the Convergence of Approximate and Regularized Policy Iteration Schemes","date":"2019-09-20","arxiv_id":"1909.09621","repositories_listed":0,"syntology":null},{"url":null,"slug":"redirection-controller-using-reinforcement","title":"Redirection Controller Using Reinforcement Learning","date":"2019-09-20","arxiv_id":"1909.09505","repositories_listed":0,"syntology":null},{"url":null,"slug":"macs-deep-reinforcement-learning-based-sdn","title":"MACS: Deep Reinforcement Learning based SDN Controller Synchronization Policy Design","date":"2019-09-19","arxiv_id":"1909.09063","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-sound-interpretation-combining-sight","title":"Robot Sound Interpretation: Combining Sight and Sound in Learning-Based Control","date":"2019-09-19","arxiv_id":"1909.09172","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-function-estimation-in-markov-reward","title":"Instance-dependent $\\ell_\\infty$-bounds for policy evaluation in tabular reinforcement learning","date":"2019-09-19","arxiv_id":"1909.08749","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-two-tier-approach-to-hyper","title":"A Hierarchical Two-tier Approach to Hyper-parameter Optimization in Reinforcement Learning","date":"2019-09-18","arxiv_id":"1909.08332","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-human-centered-data-driven-planner-actor","title":"A Human-Centered Data-Driven Planner-Actor-Critic Architecture via Logic Programming","date":"2019-09-18","arxiv_id":"1909.09209","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-lane-change-decision-making-using","title":"Automated Lane Change Decision Making using Deep Reinforcement Learning in Dynamic and Uncertain Highway Environment","date":"2019-09-18","arxiv_id":"1909.11538","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepgait-planning-and-control-of-quadrupedal","title":"DeepGait: Planning and Control of Quadrupedal Gaits using Deep Reinforcement Learning","date":"2019-09-18","arxiv_id":"1909.08399","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-aware-computation-offloading-in","title":"Dependency-Aware Computation Offloading in Mobile Edge Computing: A Reinforcement Learning Approach","date":"2019-09-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-opponent-modeling-via-adversarial","title":"Robust Opponent Modeling via Adversarial Ensemble Reinforcement Learning in Asymmetric Imperfect-Information Games","date":"2019-09-18","arxiv_id":"1909.08735","repositories_listed":0,"syntology":null},{"url":null,"slug":"segregation-dynamics-with-reinforcement","title":"Segregation Dynamics with Reinforcement Learning and Agent Based Modeling","date":"2019-09-18","arxiv_id":"1909.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-tracking-by-means-of-deep","title":"Visual Tracking by means of Deep Reinforcement Learning and an Expert Demonstrator","date":"2019-09-18","arxiv_id":"1909.08487","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-tracking-prediction-and-decision","title":"A Review of Tracking, Prediction and Decision Making Methods for Autonomous Driving","date":"2019-09-17","arxiv_id":"1909.07707","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-feature-training-for","title":"Adversarial Feature Training for Generalizable Robotic Visuomotor Control","date":"2019-09-17","arxiv_id":"1909.07745","repositories_listed":0,"syntology":null},{"url":null,"slug":"attraction-repulsion-actor-critic-for","title":"Attraction-Repulsion Actor-Critic for Continuous Control Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.07543","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-length-control-neural-encoder","title":"Controllable Length Control Neural Encoder-Decoder via Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.09492","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-black-box-adversarial-examples-for","title":"Generating Black-Box Adversarial Examples for Text Classifiers Using a Deep Reinforced Model","date":"2019-09-17","arxiv_id":"1909.07873","repositories_listed":0,"syntology":null},{"url":null,"slug":"stock-market-microstructure-inference-via","title":"Stock market microstructure inference via multi-agent reinforcement learning","date":"2019-09-17","arxiv_id":"1909.07748","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-task-driven","title":"Selective Network Discovery via Deep Reinforcement Learning on Embedded Spaces","date":"2019-09-16","arxiv_id":"1909.07294","repositories_listed":0,"syntology":null},{"url":null,"slug":"job-scheduling-on-data-centers-with-deep","title":"Data Centers Job Scheduling with Deep Reinforcement Learning","date":"2019-09-16","arxiv_id":"1909.07820","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-human-domain-knowledge-to-model-an","title":"Leveraging human Domain Knowledge to model an empirical Reward function for a Reinforcement Learning problem","date":"2019-09-16","arxiv_id":"1909.07116","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-sim-to-real","title":"Meta Reinforcement Learning for Sim-to-real Domain Adaptation","date":"2019-09-16","arxiv_id":"1909.12906","repositories_listed":0,"syntology":null}],"record_sha256":"38f761556fa862dcafc834387d4a6f6be86e07952b92f5b38b83460a38e33c54","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}