{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/125","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":125,"pages_in_order":132,"rows_per_page":100,"rows":[12401,12500],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/124","next":"/task/reinforcement-learning/papers/126","papers":[{"url":null,"slug":"feature-engineering-for-predictive-modeling","title":"Feature Engineering for Predictive Modeling using Reinforcement Learning","date":"2017-09-21","arxiv_id":"1709.07150","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-communication-protocols-for-learning","title":"Local Communication Protocols for Learning Complex Swarm Behaviors with Deep Reinforcement Learning","date":"2017-09-21","arxiv_id":"1709.07224","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for","title":"A Deep-Reinforcement Learning Approach for Software-Defined Networking Routing Optimization","date":"2017-09-20","arxiv_id":"1709.07080","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dexterous","title":"Deep Reinforcement Learning for Dexterous Manipulation with Concept Networks","date":"2017-09-20","arxiv_id":"1709.06977","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-markov-decision-processes-with-causal","title":"Sparse Markov Decision Processes with Causal Sparse Tsallis Entropy Regularization for Reinforcement Learning","date":"2017-09-19","arxiv_id":"1709.06293","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-policy-learning-in-end-to-end","title":"Iterative Policy Learning in End-to-End Trainable Task-Oriented Neural Dialog Models","date":"2017-09-18","arxiv_id":"1709.06136","repositories_listed":0,"syntology":null},{"url":null,"slug":"n2n-learning-network-to-network-compression","title":"N2N Learning: Network to Network Compression via Policy Gradient Reinforcement Learning","date":"2017-09-18","arxiv_id":"1709.06030","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-pay-more-when-you-can-pay-less-a-joint","title":"Why Pay More When You Can Pay Less: A Joint Learning Framework for Active Feature Acquisition and Classification","date":"2017-09-18","arxiv_id":"1709.05964","repositories_listed":0,"syntology":null},{"url":null,"slug":"closing-the-loop-between-neural-network","title":"Closing the loop between neural network simulators and the OpenAI Gym","date":"2017-09-17","arxiv_id":"1709.05650","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-search-through-a3c-reinforcement","title":"Improving Search through A3C Reinforcement Learning based Conversational Agent","date":"2017-09-17","arxiv_id":"1709.05638","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-variational-auto-encoder-for","title":"Disentangled Variational Auto-Encoder for Semi-supervised Learning","date":"2017-09-15","arxiv_id":"1709.05047","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-cooling-optimization-for-green","title":"Transforming Cooling Optimization for Green Data Center via Deep Reinforcement Learning","date":"2017-09-15","arxiv_id":"1709.05077","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-state-representation-learning","title":"Unsupervised state representation learning with robotic priors: a robustness benchmark","date":"2017-09-15","arxiv_id":"1709.05185","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-extracting-a-hierarchical","title":"Autonomous Extracting a Hierarchical Structure of Tasks in Reinforcement Learning and Multi-task Reinforcement Learning","date":"2017-09-14","arxiv_id":"1709.04579","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-unknown-markov-decision-processes-a","title":"Learning Unknown Markov Decision Processes: A Thompson Sampling Approach","date":"2017-09-14","arxiv_id":"1709.04570","repositories_listed":0,"syntology":null},{"url":null,"slug":"shared-learning-enhancing-reinforcement-in-q","title":"Shared Learning : Enhancing Reinforcement in $Q$-Ensembles","date":"2017-09-14","arxiv_id":"1709.04909","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-personalized-human-ai-interaction","title":"Towards personalized human AI interaction - adapting the behavior of AI agents using neural signatures of subjective interest","date":"2017-09-14","arxiv_id":"1709.04574","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-ai-population-dynamics-with","title":"A Study of AI Population Dynamics with Million-agent Reinforcement Learning","date":"2017-09-13","arxiv_id":"1709.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-surrogate","title":"Deep Reinforcement Learning with Surrogate Agent-Environment Interface","date":"2017-09-12","arxiv_id":"1709.03942","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-exploit-or-listen-combining-human","title":"Explore, Exploit or Listen: Combining Human Feedback and Policy Model to Speed up Deep Reinforcement Learning in 3D Worlds","date":"2017-09-12","arxiv_id":"1709.03969","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-stochastic-approximation-constant-step","title":"Linear Stochastic Approximation: Constant Step-Size and Iterate Averaging","date":"2017-09-12","arxiv_id":"1709.04073","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-neural-networks-with-human","title":"Pre-training Neural Networks with Human Demonstrations for Deep Reinforcement Learning","date":"2017-09-12","arxiv_id":"1709.04083","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-quadrotor-landing-using-deep","title":"Autonomous Quadrotor Landing using Deep Reinforcement Learning","date":"2017-09-11","arxiv_id":"1709.03339","repositories_listed":0,"syntology":null},{"url":null,"slug":"mbmf-model-based-priors-for-model-free","title":"MBMF: Model-Based Priors for Model-Free Reinforcement Learning","date":"2017-09-10","arxiv_id":"1709.03153","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultimate-intelligence-part-iii-measures-of","title":"Ultimate Intelligence Part III: Measures of Intelligence, Perception and Intelligent Agents","date":"2017-09-08","arxiv_id":"1709.03879","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-chatbot","title":"A Deep Reinforcement Learning Chatbot","date":"2017-09-07","arxiv_id":"1709.02349","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximating-meta-heuristics-with-homotopic","title":"Approximating meta-heuristics with homotopic recurrent neural networks","date":"2017-09-07","arxiv_id":"1709.02194","repositories_listed":0,"syntology":null},{"url":null,"slug":"formulation-of-deep-reinforcement-learning","title":"Formulation of Deep Reinforcement Learning Architecture Toward Autonomous Driving for On-Ramp Merge","date":"2017-09-07","arxiv_id":"1709.02066","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-neural-machine-translation-with","title":"Towards Neural Machine Translation with Latent Tree Attention","date":"2017-09-06","arxiv_id":"1709.01915","repositories_listed":0,"syntology":null},{"url":null,"slug":"book-storing-algorithm-invariant-episodes-for","title":"BOOK: Storing Algorithm-Invariant Episodes for Deep Reinforcement Learning","date":"2017-09-05","arxiv_id":"1709.01308","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-query-based-generative-model-for","title":"A Unified Query-based Generative Model for Question Generation and Question Answering","date":"2017-09-04","arxiv_id":"1709.01058","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-aware-dropout-dqn-for-safe-and","title":"Agent-Aware Dropout DQN for Safe and Efficient On-line Dialogue Policy Learning","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bibi-system-description-building-with-cnns","title":"BIBI System Description: Building with CNNs and Breaking with Deep Reinforcement Learning","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-what-to-read-focused-machine-reading","title":"Learning what to read: Focused machine reading","date":"2017-09-01","arxiv_id":"1709.00149","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-autonomous-control-of-distributed","title":"Resilient Autonomous Control of Distributed Multi-agent Systems in Contested Environments","date":"2017-08-31","arxiv_id":"1708.09630","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotic-bias-of-stochastic-gradient-search","title":"Asymptotic Bias of Stochastic Gradient Search","date":"2017-08-30","arxiv_id":"1709.00291","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-and-learning-control-for-autonomous","title":"Optimal and Learning Control for Autonomous Robots","date":"2017-08-30","arxiv_id":"1708.09342","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-price-with-reference-effects","title":"Learning to Price with Reference Effects","date":"2017-08-29","arxiv_id":"1708.09020","repositories_listed":0,"syntology":null},{"url":null,"slug":"novel-sensor-scheduling-scheme-for-intruder","title":"Novel Sensor Scheduling Scheme for Intruder Tracking in Energy Efficient Sensor Networks","date":"2017-08-27","arxiv_id":"1708.08113","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-q-learning-for-minimizing-demand","title":"Multi-Agent Q-Learning for Minimizing Demand-Supply Power Deficit in Microgrids","date":"2017-08-25","arxiv_id":"1708.07732","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-mechanism-design-for-e-commerce","title":"Reinforcement Mechanism Design for e-commerce","date":"2017-08-25","arxiv_id":"1708.07607","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-function-approximation-method-for-model","title":"A Function Approximation Method for Model-based High-Dimensional Inverse Reinforcement Learning","date":"2017-08-23","arxiv_id":"1708.07738","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-pomdps-with","title":"Reinforcement Learning in POMDPs with Memoryless Options and Option-Observation Initiation Sets","date":"2017-08-22","arxiv_id":"1708.06551","repositories_listed":0,"syntology":null},{"url":null,"slug":"fake-news-in-social-networks","title":"Fake News in Social Networks","date":"2017-08-21","arxiv_id":"1708.06233","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-q-network-for-the-beer-game-a-deep","title":"A Deep Q-Network for the Beer Game: A Deep Reinforcement Learning algorithm to Solve Inventory Optimization Problems","date":"2017-08-20","arxiv_id":"1708.05924","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-a-new-3d-bin-packing-problem-with","title":"Solving a New 3D Bin Packing Problem with Deep Reinforcement Learning Method","date":"2017-08-20","arxiv_id":"1708.05930","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brief-survey-of-deep-reinforcement-learning","title":"A Brief Survey of Deep Reinforcement Learning","date":"2017-08-19","arxiv_id":"1708.05866","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-forecasting-by-imitating-dynamics-in","title":"Visual Forecasting by Imitating Dynamics in Natural Sequences","date":"2017-08-19","arxiv_id":"1708.05827","repositories_listed":0,"syntology":null},{"url":null,"slug":"ladder-a-human-level-bidding-agent-for-large","title":"LADDER: A Human-Level Bidding Agent for Large-Scale Real-Time Online Auctions","date":"2017-08-18","arxiv_id":"1708.05565","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-high","title":"Deep Reinforcement Learning for High Precision Assembly Tasks","date":"2017-08-14","arxiv_id":"1708.04033","repositories_listed":0,"syntology":null},{"url":null,"slug":"belief-tree-search-for-active-object","title":"Belief Tree Search for Active Object Recognition","date":"2017-08-13","arxiv_id":"1708.03901","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-machine-learning-approach-to-routing","title":"A Machine Learning Approach to Routing","date":"2017-08-10","arxiv_id":"1708.03074","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-aware-face-hallucination-via-deep","title":"Attention-Aware Face Hallucination via Deep Reinforcement Learning","date":"2017-08-10","arxiv_id":"1708.03132","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupled-learning-of-environment","title":"Decoupled Learning of Environment Characteristics for Safe Exploration","date":"2017-08-09","arxiv_id":"1708.02838","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-policies-for-adaptive-tracking-with","title":"Learning Policies for Adaptive Tracking with Deep Feature Cascades","date":"2017-08-09","arxiv_id":"1708.02973","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-reinforcement-learning-agents","title":"Investigating Reinforcement Learning Agents for Continuous State Space Environments","date":"2017-08-08","arxiv_id":"1708.02378","repositories_listed":0,"syntology":null},{"url":null,"slug":"gplac-generalizing-vision-based-robotic","title":"GPLAC: Generalizing Vision-Based Robotic Skills using Weakly Labeled Images","date":"2017-08-07","arxiv_id":"1708.02313","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-video-captioning-with-entailment","title":"Reinforced Video Captioning with Entailment Rewards","date":"2017-08-07","arxiv_id":"1708.02300","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-optimality-principle","title":"An Information-Theoretic Optimality Principle for Deep Reinforcement Learning","date":"2017-08-06","arxiv_id":"1708.01867","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-guided-regression-network-with-context","title":"Query-guided Regression Network with Context Policy for Phrase Grounding","date":"2017-08-04","arxiv_id":"1708.01676","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-sketching-methods-for-value","title":"Effective sketching methods for value function approximation","date":"2017-08-03","arxiv_id":"1708.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-techniques-for-outer","title":"Reinforcement learning techniques for Outer Loop Link Adaptation in 4G/5G systems","date":"2017-08-03","arxiv_id":"1708.00994","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-umd-neural-machine-translation-systems-at","title":"The UMD Neural Machine Translation Systems at WMT17 Bandit Learning Task","date":"2017-08-03","arxiv_id":"1708.01318","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-inquiry","title":"Deep Reinforcement Learning for Inquiry Dialog Policies with Logical Formula Embeddings","date":"2017-08-02","arxiv_id":"1708.00667","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosted-fitted-q-iteration","title":"Boosted Fitted Q-Iteration","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-subtask-discovery-with-non","title":"Hierarchical Subtask Discovery With Non-Negative Matrix Factorization","date":"2017-08-01","arxiv_id":"1708.00463","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchy-through-composition-with-multitask","title":"Hierarchy Through Composition with Multitask LMDPs","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-stochastic-policy-gradients-in","title":"Improving Stochastic Policy Gradients in Continuous Control with Deep Reinforcement Learning using the Beta Distribution","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-optimizer-search-using-reinforcement","title":"Neural Optimizer Search using Reinforcement Learning","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"plan-attend-generate-character-level-neural","title":"Plan, Attend, Generate: Character-Level Neural Machine Translation with Planning","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"should-neural-network-architecture-reflect","title":"Should Neural Network Architecture Reflect Linguistic Structure?","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-to-model","title":"Using Reinforcement Learning to Model Incrementality in a Fast-Paced Dialogue Game","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"world-of-bits-an-open-domain-platform-for-web","title":"World of Bits: An Open-Domain Platform for Web-Based Agents","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"advantages-and-limitations-of-using-successor","title":"Advantages and Limitations of using Successor Features for Transfer in Reinforcement Learning","date":"2017-07-31","arxiv_id":"1708.00102","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectrum-access-in-cognitive-radio-using-a","title":"Spectrum Access In Cognitive Radio Using A Two Stage Reinforcement Learning Approach","date":"2017-07-31","arxiv_id":"1707.09792","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-learning-in-multiagent","title":"A Survey of Learning in Multiagent Environments: Dealing with Non-Stationarity","date":"2017-07-28","arxiv_id":"1707.09183","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-in-large-state","title":"Inverse Reinforcement Learning in Large State Spaces via Function Approximation","date":"2017-07-28","arxiv_id":"1707.09394","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-teach-reinforcement-learning","title":"Learning to Teach Reinforcement Learning Agents","date":"2017-07-28","arxiv_id":"1707.09079","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-load-control-of-thermostatically","title":"Direct Load Control of Thermostatically Controlled Loads Based on Sparse Observations Using Deep Reinforcement Learning","date":"2017-07-26","arxiv_id":"1707.08553","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-reinforcement-learning-exploration","title":"Guiding Reinforcement Learning Exploration Using Natural Language","date":"2017-07-26","arxiv_id":"1707.08616","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sparse-representations-in","title":"Learning Sparse Representations in Reinforcement Learning with Sparse Coding","date":"2017-07-26","arxiv_id":"1707.08316","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-machine-learning-a-classical","title":"Quantum machine learning: a classical perspective","date":"2017-07-26","arxiv_id":"1707.08561","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutual-alignment-transfer-learning","title":"Mutual Alignment Transfer Learning","date":"2017-07-25","arxiv_id":"1707.07907","repositories_listed":0,"syntology":null},{"url":null,"slug":"bellman-gradient-iteration-for-inverse","title":"Bellman Gradient Iteration for Inverse Reinforcement Learning","date":"2017-07-24","arxiv_id":"1707.07767","repositories_listed":0,"syntology":null},{"url":null,"slug":"3dcnn-dqn-rnn-a-deep-reinforcement-learning","title":"3DCNN-DQN-RNN: A Deep Reinforcement Learning Framework for Semantic Parsing of Large-scale 3D Point Clouds","date":"2017-07-21","arxiv_id":"1707.06783","repositories_listed":0,"syntology":null},{"url":null,"slug":"pragmatic-pedagogic-value-alignment","title":"Pragmatic-Pedagogic Value Alignment","date":"2017-07-20","arxiv_id":"1707.06354","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-balancing-for-statistical-spoken","title":"Reward-Balancing for Statistical Spoken Dialogue Systems using Multi-objective Reinforcement Learning","date":"2017-07-19","arxiv_id":"1707.06299","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-evaluation-of-a-q-learning","title":"Empirical evaluation of a Q-Learning Algorithm for Model-free Autonomous Soaring","date":"2017-07-18","arxiv_id":"1707.05668","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-line-building-energy-optimization-using","title":"On-line Building Energy Optimization using Deep Reinforcement Learning","date":"2017-07-18","arxiv_id":"1707.05878","repositories_listed":0,"syntology":null},{"url":null,"slug":"reverse-curriculum-generation-for","title":"Reverse Curriculum Generation for Reinforcement Learning","date":"2017-07-17","arxiv_id":"1707.05300","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-as-online-decision-making-learning-a","title":"Tracking as Online Decision-Making: Learning a Policy from Streaming Videos with Reinforcement Learning","date":"2017-07-17","arxiv_id":"1707.04991","repositories_listed":0,"syntology":null},{"url":null,"slug":"freeway-merging-in-congested-traffic-based-on","title":"Freeway Merging in Congested Traffic based on Multipolicy Decision Making with Passive Actor Critic","date":"2017-07-14","arxiv_id":"1707.04489","repositories_listed":0,"syntology":null},{"url":null,"slug":"distral-robust-multitask-reinforcement","title":"Distral: Robust Multitask Reinforcement Learning","date":"2017-07-13","arxiv_id":"1707.04175","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoencoder-augmented-neuroevolution-for","title":"Autoencoder-augmented Neuroevolution for Visual Doom Playing","date":"2017-07-12","arxiv_id":"1707.03902","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastest-convergence-for-q-learning","title":"Fastest Convergence for Q-learning","date":"2017-07-12","arxiv_id":"1707.03770","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-macromanagement-in-starcraft-from","title":"Learning Macromanagement in StarCraft from Replays using Deep Learning","date":"2017-07-12","arxiv_id":"1707.03743","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-for-self-organizing-networks","title":"Deep Q-Learning for Self-Organizing Networks Fault Management and Radio Performance Improvement","date":"2017-07-10","arxiv_id":"1707.02329","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-attention","title":"Deep Reinforcement Learning Attention Selection for Person Re-Identification","date":"2017-07-10","arxiv_id":"1707.02785","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-algorithm-for-volte-closed-loop","title":"Q-Learning Algorithm for VoLTE Closed-Loop Power Control in Indoor Small Cells","date":"2017-07-10","arxiv_id":"1707.03269","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-design-games-strategic","title":"Learning to Design Games: Strategic Environments in Reinforcement Learning","date":"2017-07-05","arxiv_id":"1707.01310","repositories_listed":0,"syntology":null}],"record_sha256":"18909199f1a68caeb848aa10c63eb5f02828b3a1bda24defb8b239fe516b42df","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}