{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/121","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":121,"pages_in_order":135,"rows_per_page":100,"rows":[12001,12100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/120","next":"/task/reinforcement-learning-2/papers/122","papers":[{"url":null,"slug":"learning-to-control-visual-abstractions-for","title":"Learning to Control Visual Abstractions for Structured Exploration in Deep Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-decompose-compound-questions-with","title":"Learning to Decompose Compound Questions with Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-progressively-plan","title":"Learning to Progressively Plan","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reinforcement-learn-by-imitation","title":"Learning to Reinforcement Learn by Imitation","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-solve-circuit-sat-an-unsupervised","title":"Learning To Solve Circuit-SAT: An Unsupervised Differentiable Approach","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"m3rl-mind-aware-multi-agent-management","title":"M^3RL: Mind-aware Multi-agent Management Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-the-long-term-future-in-model-based","title":"Modeling the Long Term Future in Model-Based Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-malware-control-with-deep","title":"NEURAL MALWARE CONTROL WITH DEEP REINFORCEMENT LEARNING","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rating-continuous-actions-in-spatial-multi","title":"Rating Continuous Actions in Spatial Multi-Agent Problems","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-policy-learning-in-multi","title":"Sample-efficient policy learning in multi-agent Reinforcement Learning via meta-learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simile-introducing-sequential-information","title":"SIMILE: Introducing Sequential Information towards More Effective Imitation Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uncovering-surprising-behaviors-in","title":"Uncovering Surprising Behaviors in Reinforcement Learning via Worst-case Analysis","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-generalizing-alphago-zero","title":"Understanding & Generalizing AlphaGo Zero","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"visceral-machines-reinforcement-learning-with-1","title":"Visceral Machines: Reinforcement Learning with Intrinsic Physiological Rewards","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-imagination-for-sample","title":"Generative Adversarial Imagination for Sample Efficient Deep Reinforcement Learning","date":"2019-04-30","arxiv_id":"1904.13255","repositories_listed":0,"syntology":null},{"url":null,"slug":"argus-smartphone-enabled-human-cooperation","title":"Argus: Smartphone-enabled Human Cooperation via Multi-Agent Reinforcement Learning for Disaster Situational Awareness","date":"2019-04-29","arxiv_id":"1906.03037","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-scheduler-for-vehicle","title":"Reinforcement Learning Scheduler for Vehicle-to-Vehicle Communications Outside Coverage","date":"2019-04-29","arxiv_id":"1904.12653","repositories_listed":0,"syntology":null},{"url":null,"slug":"arbitrage-of-energy-storage-in-electricity","title":"Arbitrage of Energy Storage in Electricity Markets with Deep Reinforcement Learning","date":"2019-04-28","arxiv_id":"1904.12232","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-training-autonomous-driving-agent","title":"Self Training Autonomous Driving Agent","date":"2019-04-26","arxiv_id":"1904.12738","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-optimal-1","title":"Deep Reinforcement Learning for Optimal Critical Care Pain Management with Morphine using Dueling Double-Deep Q Networks","date":"2019-04-25","arxiv_id":"1904.11115","repositories_listed":0,"syntology":null},{"url":null,"slug":"ray-interference-a-source-of-plateaus-in-deep","title":"Ray Interference: a Source of Plateaus in Deep Reinforcement Learning","date":"2019-04-25","arxiv_id":"1904.11455","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-voltage-control-for-grid-operation","title":"Autonomous Voltage Control for Grid Operation Using Deep Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.10597","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognitive-radar-using-reinforcement-learning","title":"Cognitive Radar Using Reinforcement Learning in Automotive Applications","date":"2019-04-24","arxiv_id":"1904.10739","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-neural-networks-in-reinforcement","title":"Evolving Neural Networks in Reinforcement Learning by means of UMDAc","date":"2019-04-24","arxiv_id":"1904.10932","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-you-act-tells-a-lot-privacy-leakage","title":"How You Act Tells a Lot: Privacy-Leakage Attack on Deep Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.11082","repositories_listed":0,"syntology":null},{"url":null,"slug":"driving-decision-and-control-for-autonomous","title":"Driving Decision and Control for Autonomous Lane Change based on Deep Reinforcement Learning","date":"2019-04-23","arxiv_id":"1904.10171","repositories_listed":0,"syntology":null},{"url":null,"slug":"compression-and-localization-in-reinforcement","title":"Compression and Localization in Reinforcement Learning for ATARI Games","date":"2019-04-20","arxiv_id":"1904.09489","repositories_listed":0,"syntology":null},{"url":null,"slug":"190501357","title":"Teaching on a Budget in Multi-Agent Deep Reinforcement Learning","date":"2019-04-19","arxiv_id":"1905.01357","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoding-molecular-graph-embeddings-with","title":"Decoding Molecular Graph Embeddings with Reinforcement Learning","date":"2019-04-18","arxiv_id":"1904.08915","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-interactive-reinforcement-agent","title":"Improving Interactive Reinforcement Agent Planning with Human Demonstration","date":"2019-04-18","arxiv_id":"1904.08621","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-meaning-semiotics-within-predictive","title":"Making Meaning: Semiotics Within Predictive Knowledge Architectures","date":"2019-04-18","arxiv_id":"1904.09023","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-is-a-prediction-knowledge","title":"When is a Prediction Knowledge?","date":"2019-04-18","arxiv_id":"1904.09024","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-game-theoretical-framework-for-the","title":"A Game Theoretical Framework for the Evaluation of Unmanned Aircraft Systems Airspace Integration Concepts","date":"2019-04-17","arxiv_id":"1904.08477","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-traffic-signal-control-methods","title":"A Survey on Traffic Signal Control Methods","date":"2019-04-17","arxiv_id":"1904.08117","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-emotional","title":"Reinforcement Learning Based Emotional Editing Constraint Conversation Generation","date":"2019-04-17","arxiv_id":"1904.08061","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-3d-navigation-protocols-on-touch","title":"Learning 3D Navigation Protocols on Touch Interfaces with Cooperative Multi-Agent Reinforcement Learning","date":"2019-04-16","arxiv_id":"1904.07802","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-nested-polar-code","title":"Reinforcement Learning for Nested Polar Code Construction","date":"2019-04-16","arxiv_id":"1904.07511","repositories_listed":0,"syntology":null},{"url":null,"slug":"simion-zoo-a-workbench-for-distributed","title":"Simion Zoo: A Workbench for Distributed Experimentation with Reinforcement Learning for Continuous Control Tasks","date":"2019-04-16","arxiv_id":"1904.07817","repositories_listed":0,"syntology":null},{"url":null,"slug":"curious-ilqr-resolving-uncertainty-in-model","title":"Curious iLQR: Resolving Uncertainty in Model-based RL","date":"2019-04-15","arxiv_id":"1904.06786","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-options-with-hellinger-distance","title":"Disentangling Options with Hellinger Distance Regularizer","date":"2019-04-15","arxiv_id":"1904.06887","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-interactive-reinforcement-learning","title":"Improving interactive reinforcement learning: What makes a good teacher?","date":"2019-04-15","arxiv_id":"1904.06879","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-autonomous-braking-system","title":"Multi-Objective Autonomous Braking System using Naturalistic Dataset","date":"2019-04-15","arxiv_id":"1904.07705","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-probabilistic","title":"Reinforcement Learning with Probabilistic Guarantees for Autonomous Driving","date":"2019-04-15","arxiv_id":"1904.07189","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-survey-on-memory-based-reinforcement","title":"A Short Survey On Memory Based Reinforcement Learning","date":"2019-04-14","arxiv_id":"1904.06736","repositories_listed":0,"syntology":null},{"url":null,"slug":"dot-to-dot-achieving-structured-robotic","title":"Dot-to-Dot: Explainable Hierarchical Reinforcement Learning for Robotic Manipulation","date":"2019-04-14","arxiv_id":"1904.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-scheduling-function-design-in-sdn","title":"Effective Scheduling Function Design in SDN through Deep Reinforcement Learning","date":"2019-04-12","arxiv_id":"1904.06039","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for","title":"Model-Free Reinforcement Learning for Financial Portfolios: A Brief Survey","date":"2019-04-10","arxiv_id":"1904.04973","repositories_listed":0,"syntology":null},{"url":null,"slug":"safer-deep-rl-with-shallow-mcts-a-case-study","title":"Safer Deep RL with Shallow MCTS: A Case Study in Pommerman","date":"2019-04-10","arxiv_id":"1904.05759","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-pro-level-ai-for-real-time-fighting","title":"Creating Pro-Level AI for a Real-Time Fighting Game Using Deep Reinforcement Learning","date":"2019-04-08","arxiv_id":"1904.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"jam-me-if-you-can-defeating-jammer-with-deep","title":"\"Jam Me If You Can'': Defeating Jammer with Deep Dueling Neural Network Architecture and Ambient Backscattering Augmented Communications","date":"2019-04-08","arxiv_id":"1904.03897","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-attention-that","title":"Reinforcement Learning with Attention that Works: A Self-Supervised Approach","date":"2019-04-06","arxiv_id":"1904.03367","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-preference-actor-critic","title":"Multi-Preference Actor Critic","date":"2019-04-05","arxiv_id":"1904.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-adapting-goals-allow-transfer-of","title":"Self-Adapting Goals Allow Transfer of Predictive Models to New Tasks","date":"2019-04-04","arxiv_id":"1904.02435","repositories_listed":0,"syntology":null},{"url":null,"slug":"paintbot-a-reinforcement-learning-approach","title":"PaintBot: A Reinforcement Learning Approach for Natural Media Painting","date":"2019-04-03","arxiv_id":"1904.02201","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-left-atrial-appendage-orifice","title":"Centerline Depth World Reinforcement Learning-based Left Atrial Appendage Orifice Localization","date":"2019-04-02","arxiv_id":"1904.01241","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-and-visualizing-weaknesses-of-deep","title":"Finding and Visualizing Weaknesses of Deep Reinforcement Learning Agents","date":"2019-04-02","arxiv_id":"1904.01318","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-power-control-for-large-energy","title":"Distributed Power Control for Large Energy Harvesting Networks: A Multi-Agent Deep Reinforcement Learning Approach","date":"2019-04-01","arxiv_id":"1904.00601","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-meta-policy-search","title":"Guided Meta-Policy Search","date":"2019-04-01","arxiv_id":"1904.00956","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-reinforcement-1","title":"Cooperative Multi-Agent Reinforcement Learning Framework for Scalping Trading","date":"2019-03-31","arxiv_id":"1904.00441","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-control-for-wireless-vbr-video","title":"Power Control for Wireless VBR Video Streaming: From Optimization to Reinforcement Learning","date":"2019-03-31","arxiv_id":"1904.00327","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-robust-adversarial-reinforcement","title":"Risk Averse Robust Adversarial Reinforcement Learning","date":"2019-03-31","arxiv_id":"1904.00511","repositories_listed":0,"syntology":null},{"url":null,"slug":"lane-change-decision-making-through-deep","title":"Lane Change Decision-making through Deep Reinforcement Learning with Rule-based Constraints","date":"2019-03-30","arxiv_id":"1904.00231","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-highway-driving-using-deep","title":"Autonomous Highway Driving using Deep Reinforcement Learning","date":"2019-03-29","arxiv_id":"1904.00035","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-reinforcement-learning-with","title":"Improved Reinforcement Learning with Curriculum","date":"2019-03-29","arxiv_id":"1903.12328","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-good-representation-via-continuous","title":"Learning Good Representation via Continuous Attention","date":"2019-03-29","arxiv_id":"1903.12344","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-data-detection-for-mimo-systems-with","title":"Robust Data Detection for MIMO Systems with One-Bit ADCs: A Reinforcement Learning Approach","date":"2019-03-29","arxiv_id":"1903.12546","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-brain-inspired-system-deep-recurrent","title":"Towards Brain-inspired System: Deep Recurrent Reinforcement Learning for Simulated Self-driving Agent","date":"2019-03-29","arxiv_id":"1903.12517","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-trajectory-optimization-with","title":"Regularizing Trajectory Optimization with Denoising Autoencoders","date":"2019-03-28","arxiv_id":"1903.11981","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-dependency-measure-for","title":"Wasserstein Dependency Measure for Representation Learning","date":"2019-03-28","arxiv_id":"1903.11780","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-relation-between-maximum","title":"Understanding the Relation Between Maximum-Entropy Inverse Reinforcement Learning and Behaviour Cloning","date":"2019-03-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"failure-scenario-maker-for-rule-based-agent","title":"Failure-Scenario Maker for Rule-Based Agent using Multi-agent Adversarial Reinforcement Learning and its Application to Autonomous Driving","date":"2019-03-26","arxiv_id":"1903.10654","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-text-style","title":"Reinforcement Learning Based Text Style Transfer without Parallel Training Corpus","date":"2019-03-26","arxiv_id":"1903.10671","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-use-of-deep-autoencoders-for-efficient","title":"On the use of Deep Autoencoders for Efficient Embedded Reinforcement Learning","date":"2019-03-25","arxiv_id":"1903.10404","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-task-discovery-with-limited-supervision-a","title":"Sub-Task Discovery with Limited Supervision: A Constrained Clustering Approach","date":"2019-03-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-logic-guided-safe-reinforcement","title":"Temporal Logic Guided Safe Reinforcement Learning Using Control Barrier Functions","date":"2019-03-23","arxiv_id":"1903.09885","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hierarchical-reinforcement-learning-1","title":"Deep Hierarchical Reinforcement Learning Based Recommendations via Multi-goals Abstraction","date":"2019-03-22","arxiv_id":"1903.09374","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-reinforcement-learning-to-mere","title":"Explaining Reinforcement Learning to Mere Mortals: An Empirical Study","date":"2019-03-22","arxiv_id":"1903.09708","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-safety-in-reinforcement-learning","title":"Improving Safety in Reinforcement Learning Using Model-Based Architectures and Human Intervention","date":"2019-03-22","arxiv_id":"1903.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-reinforcement-learning-with","title":"Macro Action Reinforcement Learning with Sequence Disentanglement using Variational Autoencoder","date":"2019-03-22","arxiv_id":"1903.09366","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-regression-methods-for-reinforcement","title":"Symbolic Regression Methods for Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-off-policy-actor-critic","title":"Distributed off-Policy Actor-Critic Reinforcement Learning with Policy Consensus","date":"2019-03-21","arxiv_id":"1903.09255","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-memory-networks-for-streaming-based-1","title":"Augmented Memory Networks for Streaming-Based Active One-Shot Learning","date":"2019-03-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-classical-planning-for-adversary","title":"Single-step Options for Adversary Driving","date":"2019-03-20","arxiv_id":"1903.08606","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-promoting-deep-reinforcement","title":"Diversity-Promoting Deep Reinforcement Learning for Interactive Recommendation","date":"2019-03-19","arxiv_id":"1903.07826","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with","title":"Deep Reinforcement Learning with Decorrelation","date":"2019-03-18","arxiv_id":"1903.07765","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-hierarchy-for-learning-and","title":"Exploiting Hierarchy for Learning and Transfer in KL-regularized RL","date":"2019-03-18","arxiv_id":"1903.07438","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-proposals-for-sequential-importance","title":"Learning proposals for sequential importance samplers using reinforced variational inference","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-query-reformulation-challenges","title":"Multi-agent query reformulation: Challenges and the role of diversity","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-for-autonomous","title":"Robust Reinforcement Learning for Autonomous Driving","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"online-antenna-tuning-in-heterogeneous","title":"Online Antenna Tuning in Heterogeneous Cellular Networks with Deep Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-distillation-and-value-matching-in","title":"Policy Distillation and Value Matching in Multiagent Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06592","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-markov-decision-processes-using","title":"No-regret Exploration in Contextual Reinforcement Learning","date":"2019-03-14","arxiv_id":"1903.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-applications-of-bootstrap-in-continuous","title":"On Applications of Bootstrap in Continuous Space Reinforcement Learning","date":"2019-03-14","arxiv_id":"1903.05803","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-reinforcement-learning-based-local","title":"Effective reinforcement learning based local search for the maximum k-plex problem","date":"2019-03-13","arxiv_id":"1903.05537","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-abstraction-for-reinforcement","title":"Resource Abstraction for Reinforcement Learning in Multiagent Congestion Problems","date":"2019-03-13","arxiv_id":"1903.05431","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-design-through-deep","title":"Task-oriented Design through Deep Reinforcement Learning","date":"2019-03-13","arxiv_id":"1903.05271","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-optimization-for-unknown","title":"Trajectory Optimization for Unknown Constrained Systems using Reinforcement Learning","date":"2019-03-13","arxiv_id":"1903.05751","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-reinforcement-learning-for","title":"A Review of Reinforcement Learning for Autonomous Building Energy Management","date":"2019-03-12","arxiv_id":"1903.05196","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-with-1","title":"Deep Multi-Agent Reinforcement Learning with Discrete-Continuous Hybrid Action Spaces","date":"2019-03-12","arxiv_id":"1903.04959","repositories_listed":0,"syntology":null}],"record_sha256":"ce1a035e03eeb7004acb3b68cf5d3baeb38ec358eaeb8ed19628d9d3a4856e0e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}