{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/143","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":143,"pages_in_order":152,"rows_per_page":100,"rows":[14201,14300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/142","next":"/task/reinforcement-learning-1/papers/144","papers":[{"url":null,"slug":"interactive-reinforcement-learning-with","title":"Interactive Reinforcement Learning with Dynamic Reuse of Prior Knowledge from Human/Agent's Demonstration","date":"2018-05-11","arxiv_id":"1805.04493","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-grammar-and-reinforcement-learning","title":"Leveraging Grammar and Reinforcement Learning for Neural Program Synthesis","date":"2018-05-11","arxiv_id":"1805.04276","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-optimal","title":"Deep Reinforcement Learning for Optimal Control of Space Heating","date":"2018-05-10","arxiv_id":"1805.03777","repositories_listed":0,"syntology":null},{"url":null,"slug":"discourse-aware-neural-rewards-for-coherent","title":"Discourse-Aware Neural Rewards for Coherent Text Generation","date":"2018-05-10","arxiv_id":"1805.03766","repositories_listed":0,"syntology":null},{"url":null,"slug":"metatrace-online-step-size-tuning-by-meta","title":"Metatrace Actor-Critic: Online Step-size Tuning by Meta-gradient Descent for Reinforcement Learning Control","date":"2018-05-10","arxiv_id":"1805.04514","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-hierarchical-reinforcement","title":"Multimodal Hierarchical Reinforcement Learning Policy for Task-Oriented Visual Dialog","date":"2018-05-08","arxiv_id":"1805.03257","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-page-wise","title":"Deep Reinforcement Learning for Page-wise Recommendations","date":"2018-05-07","arxiv_id":"1805.02343","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-machine-translation-with","title":"Multimodal Machine Translation with Reinforcement Learning","date":"2018-05-07","arxiv_id":"1805.02356","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-and-learning-with-stochastic-action","title":"Planning and Learning with Stochastic Action Sets","date":"2018-05-07","arxiv_id":"1805.02363","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-parsimonious-ensembles-using","title":"Developing parsimonious ensembles using ensemble diversity within a reinforcement learning framework","date":"2018-05-05","arxiv_id":"1805.02103","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-by-distributional-reinforcement","title":"Exploration by Distributional Reinforcement Learning","date":"2018-05-04","arxiv_id":"1805.01907","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deep-reinforcement-learning-for","title":"Robust Deep Reinforcement Learning for Security and Safety in Autonomous Vehicle Systems","date":"2018-05-02","arxiv_id":"1805.00983","repositories_listed":0,"syntology":null},{"url":null,"slug":"falsification-of-cyber-physical-systems-using","title":"Falsification of Cyber-Physical Systems Using Deep Reinforcement Learning","date":"2018-05-01","arxiv_id":"1805.00200","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-log-optimal-strategy-with","title":"Robust Log-Optimal Strategy with Reinforcement Learning","date":"2018-05-01","arxiv_id":"1805.00205","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-interpretable-fuzzy-controllers","title":"Generating Interpretable Fuzzy Controllers using Particle Swarm Optimization and Genetic Programming","date":"2018-04-29","arxiv_id":"1804.10960","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-experienced-anomaly-detector-through","title":"Towards Experienced Anomaly Detector through Reinforcement Learning","date":"2018-04-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiment-adaptive-end-to-end-dialog-systems","title":"Sentiment Adaptive End-to-End Dialog Systems","date":"2018-04-28","arxiv_id":"1804.10731","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-to-acquire","title":"Deep Reinforcement Learning to Acquire Navigation Skills for Wheel-Legged Robots in Complex Environments","date":"2018-04-27","arxiv_id":"1804.10500","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-categorization-for-computationally","title":"Action Categorization for Computationally Improved Task Learning and Planning","date":"2018-04-26","arxiv_id":"1804.09856","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiagent-soft-q-learning","title":"Multiagent Soft Q-Learning","date":"2018-04-25","arxiv_id":"1804.09817","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-projective-simulation-in","title":"Benchmarking projective simulation in navigation problems","date":"2018-04-23","arxiv_id":"1804.08607","repositories_listed":0,"syntology":null},{"url":null,"slug":"mqgrad-reinforcement-learning-of-gradient","title":"MQGrad: Reinforcement Learning of Gradient Quantization in Parameter Server","date":"2018-04-22","arxiv_id":"1804.08066","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-extraction-with-generative-adversarial","title":"Event Extraction with Generative Adversarial Imitation Learning","date":"2018-04-21","arxiv_id":"1804.07881","repositories_listed":0,"syntology":null},{"url":null,"slug":"peorl-integrating-symbolic-planning-and","title":"PEORL: Integrating Symbolic Planning and Hierarchical Reinforcement Learning for Robust Decision-Making","date":"2018-04-20","arxiv_id":"1804.07779","repositories_listed":0,"syntology":null},{"url":null,"slug":"cell-selection-with-deep-reinforcement","title":"Cell Selection with Deep Reinforcement Learning in Sparse Mobile Crowdsensing","date":"2018-04-19","arxiv_id":"1804.07047","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-controllable-and-uncontrollable","title":"Disentangling Controllable and Uncontrollable Factors of Variation by Interacting with the World","date":"2018-04-19","arxiv_id":"1804.06955","repositories_listed":0,"syntology":null},{"url":"/paper/learning-to-extract-coherent-summary-via-deep","slug":"learning-to-extract-coherent-summary-via-deep","title":"Learning to Extract Coherent Summary via Deep Reinforcement Learning","date":"2018-04-19","arxiv_id":"1804.07036","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-vehicles-behavior-decision-making","title":"Automated vehicle's behavior decision making using deep reinforcement learning and high-fidelity simulation environment","date":"2018-04-17","arxiv_id":"1804.06264","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-linear-quadratic-control-via","title":"Model-Free Linear Quadratic Control via Reduction to Expert Prediction","date":"2018-04-17","arxiv_id":"1804.06021","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-improving-deep-reinforcement-learning-for-1","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2018-04-17","arxiv_id":"1804.06309","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-how-to-self-learn-enhancing-self","title":"Learning How to Self-Learn: Enhancing Self-Training Using Neural Reinforcement Learning","date":"2018-04-16","arxiv_id":"1804.05734","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-augmentation-transformations-for-risk","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","date":"2018-04-16","arxiv_id":"1804.05950","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-dual-view-deep-agent","title":"Robust Dual View Deep Agent","date":"2018-04-13","arxiv_id":"1804.05120","repositories_listed":0,"syntology":null},{"url":null,"slug":"distort-and-recover-color-enhancement-using","title":"Distort-and-Recover: Color Enhancement using Deep Reinforcement Learning","date":"2018-04-12","arxiv_id":"1804.04450","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-based-aggregation-and-deep","title":"Feature-Based Aggregation and Deep Reinforcement Learning: A Survey and Some New Implementations","date":"2018-04-12","arxiv_id":"1804.04577","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-query-evaluations-using","title":"Optimizing Query Evaluations using Reinforcement Learning for Web Search","date":"2018-04-12","arxiv_id":"1804.04410","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-successor-representations-for","title":"Universal Successor Representations for Transfer Reinforcement Learning","date":"2018-04-11","arxiv_id":"1804.03758","repositories_listed":0,"syntology":null},{"url":null,"slug":"binary-space-partitioning-as-intrinsic-reward","title":"Binary Space Partitioning as Intrinsic Reward","date":"2018-04-10","arxiv_id":"1804.03611","repositories_listed":0,"syntology":null},{"url":null,"slug":"outline-objects-using-deep-reinforcement","title":"Outline Objects using Deep Reinforcement Learning","date":"2018-04-10","arxiv_id":"1804.04603","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalization-of-health-interventions-using","title":"A clustering-based reinforcement learning approach for tailored personalization of e-Health interventions","date":"2018-04-10","arxiv_id":"1804.03592","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-space-policies-for-hierarchical","title":"Latent Space Policies for Hierarchical Reinforcement Learning","date":"2018-04-09","arxiv_id":"1804.02808","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-modular-reinforcement-learning","title":"Hierarchical Modular Reinforcement Learning Method and Knowledge Acquisition of State-Action Rule for Multi-target Problem","date":"2018-04-08","arxiv_id":"1804.02698","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-sentiment-for-sequence-to-sequence","title":"Scalable Sentiment for Sequence-to-sequence Chatbot Response with Performance Analysis","date":"2018-04-07","arxiv_id":"1804.02504","repositories_listed":0,"syntology":null},{"url":null,"slug":"programmatically-interpretable-reinforcement","title":"Programmatically Interpretable Reinforcement Learning","date":"2018-04-06","arxiv_id":"1804.02477","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-human-mixed-strategy-approach-to-deep","title":"A Human Mixed Strategy Approach to Deep Reinforcement Learning","date":"2018-04-05","arxiv_id":"1804.01874","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-maximizing-exploration-with-a","title":"Information Maximizing Exploration with a Latent Dynamics Model","date":"2018-04-04","arxiv_id":"1804.01238","repositories_listed":0,"syntology":null},{"url":null,"slug":"emorl-continuous-acoustic-emotion","title":"EmoRL: Continuous Acoustic Emotion Classification using Deep Reinforcement Learning","date":"2018-04-03","arxiv_id":"1804.04053","repositories_listed":0,"syntology":null},{"url":null,"slug":"renewal-monte-carlo-renewal-theory-based","title":"Renewal Monte Carlo: Renewal theory based reinforcement learning","date":"2018-04-03","arxiv_id":"1804.01116","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-exploration-for-mapless","title":"Curiosity-driven Exploration for Mapless Navigation with Deep Reinforcement Learning","date":"2018-04-02","arxiv_id":"1804.00456","repositories_listed":0,"syntology":null},{"url":null,"slug":"recall-traces-backtracking-models-for","title":"Recall Traces: Backtracking Models for Efficient Reinforcement Learning","date":"2018-04-02","arxiv_id":"1804.00379","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-run-challenge-synthesizing","title":"Learning to Run challenge: Synthesizing physiologically accurate motion using deep reinforcement learning","date":"2018-03-31","arxiv_id":"1804.00198","repositories_listed":0,"syntology":null},{"url":null,"slug":"snap-angle-prediction-for-360circ-panoramas","title":"Snap Angle Prediction for 360$^{\\circ}$ Panoramas","date":"2018-03-31","arxiv_id":"1804.00126","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-an-electrical-engineer-became-an","title":"How an Electrical Engineer Became an Artificial Intelligence Researcher, a Multiphase Active Contours Analysis","date":"2018-03-29","arxiv_id":"1803.11261","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-non-prehensile","title":"Reinforcement learning for non-prehensile manipulation: Transfer from simulation to physical system","date":"2018-03-28","arxiv_id":"1803.10371","repositories_listed":0,"syntology":null},{"url":"/paper/deep-communicating-agents-for-abstractive","slug":"deep-communicating-agents-for-abstractive","title":"Deep Communicating Agents for Abstractive Summarization","date":"2018-03-27","arxiv_id":"1803.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"forward-backward-reinforcement-learning","title":"Forward-Backward Reinforcement Learning","date":"2018-03-27","arxiv_id":"1803.10227","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-fair-dynamic","title":"Reinforcement Learning for Fair Dynamic Pricing","date":"2018-03-27","arxiv_id":"1803.09967","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-photonic-reinforcement-learning-by","title":"Scalable photonic reinforcement learning by time-division multiplexing of laser chaos","date":"2018-03-26","arxiv_id":"1803.09425","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-ramp-merge-maneuver-based-on","title":"Autonomous Ramp Merge Maneuver Based on Reinforcement Learning with Continuous Action Space","date":"2018-03-25","arxiv_id":"1803.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-importance-of-constraint-smoothness-for","title":"The Importance of Constraint Smoothness for Parameter Estimation in Computational Cognitive Modeling","date":"2018-03-24","arxiv_id":"1803.09018","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-learning-in-constructive","title":"Accelerating Learning in Constructive Predictive Frameworks with the Successor Representation","date":"2018-03-23","arxiv_id":"1803.09001","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-model","title":"Deep Reinforcement Learning with Model Learning and Monte Carlo Tree Search in Minecraft","date":"2018-03-22","arxiv_id":"1803.08456","repositories_listed":0,"syntology":null},{"url":null,"slug":"dop-deep-optimistic-planning-with-approximate","title":"DOP: Deep Optimistic Planning with Approximate Value Function Evaluation","date":"2018-03-22","arxiv_id":"1803.08501","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-state-representations-for-query","title":"Learning State Representations for Query Optimization with Deep Reinforcement Learning","date":"2018-03-22","arxiv_id":"1803.08604","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robotic-assembly-from-cad","title":"Learning Robotic Assembly from CAD","date":"2018-03-20","arxiv_id":"1803.07635","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-latent","title":"Meta Reinforcement Learning with Latent Variable Gaussian Processes","date":"2018-03-20","arxiv_id":"1803.07551","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-gradient-deep-q-learning","title":"Natural Gradient Deep Q-learning","date":"2018-03-20","arxiv_id":"1803.07482","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-sponsored-search-ranking-strategy","title":"Optimizing Sponsored Search Ranking Strategy by Deep Reinforcement Learning","date":"2018-03-20","arxiv_id":"1803.07347","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduction-for-policy-gradient-with","title":"Variance Reduction for Policy Gradient with Action-Dependent Factorized Baselines","date":"2018-03-20","arxiv_id":"1803.07246","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-text-generation-past-present-and","title":"Neural Text Generation: Past, Present and Beyond","date":"2018-03-15","arxiv_id":"1803.07133","repositories_listed":0,"syntology":null},{"url":null,"slug":"rearrangement-with-nonprehensile-manipulation","title":"Rearrangement with Nonprehensile Manipulation Using Deep Reinforcement Learning","date":"2018-03-15","arxiv_id":"1803.05752","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-speed-and-lane-change-decision","title":"Automated Speed and Lane Change Decision Making using Deep Reinforcement Learning","date":"2018-03-14","arxiv_id":"1803.10056","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-with-concurrent-actions-in","title":"Imitation Learning with Concurrent Actions in 3D Games","date":"2018-03-14","arxiv_id":"1803.05402","repositories_listed":0,"syntology":null},{"url":null,"slug":"measurement-based-adaptation-protocol-with-1","title":"Measurement-based adaptation protocol with quantum reinforcement learning","date":"2018-03-14","arxiv_id":"1803.05340","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-reinforcement-learning-with-monte","title":"Active Reinforcement Learning with Monte-Carlo Tree Search","date":"2018-03-13","arxiv_id":"1803.04926","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning","title":"Hierarchical Reinforcement Learning: Approximating Optimal Discounted TSP Using Local Policies","date":"2018-03-13","arxiv_id":"1803.04674","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-with-meta-policy-gradient","title":"Learning to Explore with Meta-Policy Gradient","date":"2018-03-13","arxiv_id":"1803.05044","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-search-in-continuous-action-domains-an","title":"Policy Search in Continuous Action Domains: an Overview","date":"2018-03-13","arxiv_id":"1803.04706","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-robust-actor-critic-policy-gradient","title":"Soft-Robust Actor-Critic Policy-Gradient","date":"2018-03-11","arxiv_id":"1803.04848","repositories_listed":0,"syntology":null},{"url":null,"slug":"kickstarting-deep-reinforcement-learning","title":"Kickstarting Deep Reinforcement Learning","date":"2018-03-10","arxiv_id":"1803.03835","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-objective-deep-reinforcement-learning","title":"A Multi-Objective Deep Reinforcement Learning Framework","date":"2018-03-08","arxiv_id":"1803.02965","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepcas-a-deep-reinforcement-learning","title":"DeepCAS: A Deep Reinforcement Learning Algorithm for Control-Aware Scheduling","date":"2018-03-08","arxiv_id":"1803.02998","repositories_listed":0,"syntology":null},{"url":null,"slug":"feudal-reinforcement-learning-for-dialogue","title":"Feudal Reinforcement Learning for Dialogue Management in Large Domains","date":"2018-03-08","arxiv_id":"1803.03232","repositories_listed":0,"syntology":null},{"url":null,"slug":"sa-iga-a-multiagent-reinforcement-learning","title":"SA-IGA: A Multiagent Reinforcement Learning Method Towards Socially Optimal Outcomes","date":"2018-03-08","arxiv_id":"1803.03021","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brandom-ian-view-of-reinforcement-learning","title":"A Brandom-ian view of Reinforcement Learning towards strong-AI","date":"2018-03-07","arxiv_id":"1803.02912","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-action-sequences-from-texts-based","title":"Extracting Action Sequences from Texts Based on Deep Reinforcement Learning","date":"2018-03-07","arxiv_id":"1803.02632","repositories_listed":0,"syntology":null},{"url":null,"slug":"intent-aware-multi-agent-reinforcement","title":"Intent-aware Multi-agent Reinforcement Learning","date":"2018-03-06","arxiv_id":"1803.02018","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-exposure-control-using-adaptive","title":"Personalized Exposure Control Using Adaptive Metering and Reinforcement Learning","date":"2018-03-06","arxiv_id":"1803.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothed-action-value-functions-for-learning","title":"Smoothed Action Value Functions for Learning Gaussian Policies","date":"2018-03-06","arxiv_id":"1803.02348","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-aware-regret-bounds-for-undiscounted","title":"Variance-Aware Regret Bounds for Undiscounted Reinforcement Learning in MDPs","date":"2018-03-05","arxiv_id":"1803.01626","repositories_listed":0,"syntology":null},{"url":null,"slug":"oil-observational-imitation-learning","title":"OIL: Observational Imitation Learning","date":"2018-03-03","arxiv_id":"1803.01129","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-control-for-distributed-stream","title":"Model-Free Control for Distributed Stream Data Processing using Deep Reinforcement Learning","date":"2018-03-02","arxiv_id":"1803.01016","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-sponsored","title":"Deep Reinforcement Learning for Sponsored Search Real-time Bidding","date":"2018-03-01","arxiv_id":"1803.00259","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-imitation-and-reinforcement","title":"Hierarchical Imitation and Reinforcement Learning","date":"2018-03-01","arxiv_id":"1803.00590","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-via","title":"Inverse Reinforcement Learning via Nonparametric Spatio-Temporal Subgoal Modeling","date":"2018-03-01","arxiv_id":"1803.00444","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-oracle-efficient-pac-rl-with-rich","title":"On Oracle-Efficient PAC RL with Rich Observations","date":"2018-03-01","arxiv_id":"1803.00606","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cooperation-in-sequential-prisoners","title":"Towards Cooperation in Sequential Prisoner's Dilemmas: a Deep Multiagent Reinforcement Learning Approach","date":"2018-03-01","arxiv_id":"1803.00162","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-join-order","title":"Deep Reinforcement Learning for Join Order Enumeration","date":"2018-02-28","arxiv_id":"1803.00055","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-value-estimation-for-efficient","title":"Model-Based Value Estimation for Efficient Model-Free Reinforcement Learning","date":"2018-02-28","arxiv_id":"1803.00101","repositories_listed":0,"syntology":null},{"url":null,"slug":"digrad-multi-task-reinforcement-learning-with","title":"DiGrad: Multi-Task Reinforcement Learning with Shared Actions","date":"2018-02-27","arxiv_id":"1802.10463","repositories_listed":0,"syntology":null}],"record_sha256":"0b14daaf3dca8454b0783f14a0814a970da40ac6467cfedd9c1c978bfe79ccc2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}