{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/121","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":121,"pages_in_order":132,"rows_per_page":100,"rows":[12001,12100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/120","next":"/task/reinforcement-learning/papers/122","papers":[{"url":null,"slug":"multiagent-soft-q-learning","title":"Multiagent Soft Q-Learning","date":"2018-04-25","arxiv_id":"1804.09817","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-projective-simulation-in","title":"Benchmarking projective simulation in navigation problems","date":"2018-04-23","arxiv_id":"1804.08607","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-distribution-aware-sampling-for-deep-q","title":"State Distribution-aware Sampling for Deep Q-learning","date":"2018-04-23","arxiv_id":"1804.08619","repositories_listed":0,"syntology":null},{"url":null,"slug":"mqgrad-reinforcement-learning-of-gradient","title":"MQGrad: Reinforcement Learning of Gradient Quantization in Parameter Server","date":"2018-04-22","arxiv_id":"1804.08066","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-extraction-with-generative-adversarial","title":"Event Extraction with Generative Adversarial Imitation Learning","date":"2018-04-21","arxiv_id":"1804.07881","repositories_listed":0,"syntology":null},{"url":null,"slug":"peorl-integrating-symbolic-planning-and","title":"PEORL: Integrating Symbolic Planning and Hierarchical Reinforcement Learning for Robust Decision-Making","date":"2018-04-20","arxiv_id":"1804.07779","repositories_listed":0,"syntology":null},{"url":null,"slug":"subgoal-discovery-for-hierarchical-dialogue","title":"Subgoal Discovery for Hierarchical Dialogue Policy Learning","date":"2018-04-20","arxiv_id":"1804.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"cell-selection-with-deep-reinforcement","title":"Cell Selection with Deep Reinforcement Learning in Sparse Mobile Crowdsensing","date":"2018-04-19","arxiv_id":"1804.07047","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-controllable-and-uncontrollable","title":"Disentangling Controllable and Uncontrollable Factors of Variation by Interacting with the World","date":"2018-04-19","arxiv_id":"1804.06955","repositories_listed":0,"syntology":null},{"url":"/paper/learning-to-extract-coherent-summary-via-deep","slug":"learning-to-extract-coherent-summary-via-deep","title":"Learning to Extract Coherent Summary via Deep Reinforcement Learning","date":"2018-04-19","arxiv_id":"1804.07036","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-and-simultaneously-removing-bias-via","title":"Modeling and Simultaneously Removing Bias via Adversarial Neural Networks","date":"2018-04-18","arxiv_id":"1804.06909","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adaptive-clipping-approach-for-proximal","title":"An Adaptive Clipping Approach for Proximal Policy Optimization","date":"2018-04-17","arxiv_id":"1804.06461","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-vehicles-behavior-decision-making","title":"Automated vehicle's behavior decision making using deep reinforcement learning and high-fidelity simulation environment","date":"2018-04-17","arxiv_id":"1804.06264","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-linear-quadratic-control-via","title":"Model-Free Linear Quadratic Control via Reduction to Expert Prediction","date":"2018-04-17","arxiv_id":"1804.06021","repositories_listed":0,"syntology":null},{"url":"/paper/multi-reward-reinforced-summarization-with","slug":"multi-reward-reinforced-summarization-with","title":"Multi-Reward Reinforced Summarization with Saliency and Entailment","date":"2018-04-17","arxiv_id":"1804.06451","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-improving-deep-reinforcement-learning-for-1","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2018-04-17","arxiv_id":"1804.06309","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-how-to-self-learn-enhancing-self","title":"Learning How to Self-Learn: Enhancing Self-Training Using Neural Reinforcement Learning","date":"2018-04-16","arxiv_id":"1804.05734","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modality-sensor-data-classification","title":"Multi-modality Sensor Data Classification with Selective Attention","date":"2018-04-16","arxiv_id":"1804.05493","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-convergence-of-gradient-based-learning","title":"On Gradient-Based Learning in Continuous Games","date":"2018-04-16","arxiv_id":"1804.05464","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-augmentation-transformations-for-risk","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","date":"2018-04-16","arxiv_id":"1804.05950","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-deep-sketch-abstraction","title":"Learning Deep Sketch Abstraction","date":"2018-04-13","arxiv_id":"1804.04804","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-dual-view-deep-agent","title":"Robust Dual View Deep Agent","date":"2018-04-13","arxiv_id":"1804.05120","repositories_listed":0,"syntology":null},{"url":null,"slug":"distort-and-recover-color-enhancement-using","title":"Distort-and-Recover: Color Enhancement using Deep Reinforcement Learning","date":"2018-04-12","arxiv_id":"1804.04450","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-based-aggregation-and-deep","title":"Feature-Based Aggregation and Deep Reinforcement Learning: A Survey and Some New Implementations","date":"2018-04-12","arxiv_id":"1804.04577","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-query-evaluations-using","title":"Optimizing Query Evaluations using Reinforcement Learning for Web Search","date":"2018-04-12","arxiv_id":"1804.04410","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-dynamics-models-for-adaptive","title":"Personalized Dynamics Models for Adaptive Assistive Navigation Systems","date":"2018-04-11","arxiv_id":"1804.04118","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-successor-representations-for","title":"Universal Successor Representations for Transfer Reinforcement Learning","date":"2018-04-11","arxiv_id":"1804.03758","repositories_listed":0,"syntology":null},{"url":null,"slug":"binary-space-partitioning-as-intrinsic-reward","title":"Binary Space Partitioning as Intrinsic Reward","date":"2018-04-10","arxiv_id":"1804.03611","repositories_listed":0,"syntology":null},{"url":null,"slug":"outline-objects-using-deep-reinforcement","title":"Outline Objects using Deep Reinforcement Learning","date":"2018-04-10","arxiv_id":"1804.04603","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalization-of-health-interventions-using","title":"A clustering-based reinforcement learning approach for tailored personalization of e-Health interventions","date":"2018-04-10","arxiv_id":"1804.03592","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-space-policies-for-hierarchical","title":"Latent Space Policies for Hierarchical Reinforcement Learning","date":"2018-04-09","arxiv_id":"1804.02808","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-with-value-function","title":"Policy Gradient With Value Function Approximation For Collective Multiagent Planning","date":"2018-04-09","arxiv_id":"1804.02884","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-modular-reinforcement-learning","title":"Hierarchical Modular Reinforcement Learning Method and Knowledge Acquisition of State-Action Rule for Multi-target Problem","date":"2018-04-08","arxiv_id":"1804.02698","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-sentiment-for-sequence-to-sequence","title":"Scalable Sentiment for Sequence-to-sequence Chatbot Response with Performance Analysis","date":"2018-04-07","arxiv_id":"1804.02504","repositories_listed":0,"syntology":null},{"url":null,"slug":"programmatically-interpretable-reinforcement","title":"Programmatically Interpretable Reinforcement Learning","date":"2018-04-06","arxiv_id":"1804.02477","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-qosqoe-aware","title":"Reinforcement Learning based QoS/QoE-aware Service Function Chaining in Software-Driven 5G Slices","date":"2018-04-06","arxiv_id":"1804.02099","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-human-mixed-strategy-approach-to-deep","title":"A Human Mixed Strategy Approach to Deep Reinforcement Learning","date":"2018-04-05","arxiv_id":"1804.01874","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-maximizing-exploration-with-a","title":"Information Maximizing Exploration with a Latent Dynamics Model","date":"2018-04-04","arxiv_id":"1804.01238","repositories_listed":0,"syntology":null},{"url":null,"slug":"emorl-continuous-acoustic-emotion","title":"EmoRL: Continuous Acoustic Emotion Classification using Deep Reinforcement Learning","date":"2018-04-03","arxiv_id":"1804.04053","repositories_listed":0,"syntology":null},{"url":null,"slug":"renewal-monte-carlo-renewal-theory-based","title":"Renewal Monte Carlo: Renewal theory based reinforcement learning","date":"2018-04-03","arxiv_id":"1804.01116","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-exploration-for-mapless","title":"Curiosity-driven Exploration for Mapless Navigation with Deep Reinforcement Learning","date":"2018-04-02","arxiv_id":"1804.00456","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-to-learn-visual-saliency-the-rl-iac","title":"Exploring to learn visual saliency: The RL-IAC approach","date":"2018-04-02","arxiv_id":"1804.00435","repositories_listed":0,"syntology":null},{"url":null,"slug":"recall-traces-backtracking-models-for","title":"Recall Traces: Backtracking Models for Efficient Reinforcement Learning","date":"2018-04-02","arxiv_id":"1804.00379","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-intelligent-vehicular-networks-a","title":"Toward Intelligent Vehicular Networks: A Machine Learning Framework","date":"2018-04-01","arxiv_id":"1804.00338","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-run-challenge-synthesizing","title":"Learning to Run challenge: Synthesizing physiologically accurate motion using deep reinforcement learning","date":"2018-03-31","arxiv_id":"1804.00198","repositories_listed":0,"syntology":null},{"url":null,"slug":"snap-angle-prediction-for-360circ-panoramas","title":"Snap Angle Prediction for 360$^{\\circ}$ Panoramas","date":"2018-03-31","arxiv_id":"1804.00126","repositories_listed":0,"syntology":null},{"url":null,"slug":"observer-based-adaptive-optimal-output","title":"Observer-based Adaptive Optimal Output Containment Control problem of Linear Heterogeneous Multi-agent Systems with Relative Output Measurements","date":"2018-03-30","arxiv_id":"1803.11411","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-an-electrical-engineer-became-an","title":"How an Electrical Engineer Became an Artificial Intelligence Researcher, a Multiphase Active Contours Analysis","date":"2018-03-29","arxiv_id":"1803.11261","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-based-training-framework-for","title":"Actor-Critic based Training Framework for Abstractive Summarization","date":"2018-03-28","arxiv_id":"1803.11070","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-non-prehensile","title":"Reinforcement learning for non-prehensile manipulation: Transfer from simulation to physical system","date":"2018-03-28","arxiv_id":"1803.10371","repositories_listed":0,"syntology":null},{"url":"/paper/deep-communicating-agents-for-abstractive","slug":"deep-communicating-agents-for-abstractive","title":"Deep Communicating Agents for Abstractive Summarization","date":"2018-03-27","arxiv_id":"1803.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"forward-backward-reinforcement-learning","title":"Forward-Backward Reinforcement Learning","date":"2018-03-27","arxiv_id":"1803.10227","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-fair-dynamic","title":"Reinforcement Learning for Fair Dynamic Pricing","date":"2018-03-27","arxiv_id":"1803.09967","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-end-to-end-imitation-learning-for-model","title":"Safe end-to-end imitation learning for model predictive control","date":"2018-03-27","arxiv_id":"1803.10231","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-photonic-reinforcement-learning-by","title":"Scalable photonic reinforcement learning by time-division multiplexing of laser chaos","date":"2018-03-26","arxiv_id":"1803.09425","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-ramp-merge-maneuver-based-on","title":"Autonomous Ramp Merge Maneuver Based on Reinforcement Learning with Continuous Action Space","date":"2018-03-25","arxiv_id":"1803.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-importance-of-constraint-smoothness-for","title":"The Importance of Constraint Smoothness for Parameter Estimation in Computational Cognitive Modeling","date":"2018-03-24","arxiv_id":"1803.09018","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-learning-in-constructive","title":"Accelerating Learning in Constructive Predictive Frameworks with the Successor Representation","date":"2018-03-23","arxiv_id":"1803.09001","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-model","title":"Deep Reinforcement Learning with Model Learning and Monte Carlo Tree Search in Minecraft","date":"2018-03-22","arxiv_id":"1803.08456","repositories_listed":0,"syntology":null},{"url":null,"slug":"dop-deep-optimistic-planning-with-approximate","title":"DOP: Deep Optimistic Planning with Approximate Value Function Evaluation","date":"2018-03-22","arxiv_id":"1803.08501","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-state-representations-for-query","title":"Learning State Representations for Query Optimization with Deep Reinforcement Learning","date":"2018-03-22","arxiv_id":"1803.08604","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-rapidly-changing-landscape-of","title":"The Rapidly Changing Landscape of Conversational Agents","date":"2018-03-22","arxiv_id":"1803.08419","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robotic-assembly-from-cad","title":"Learning Robotic Assembly from CAD","date":"2018-03-20","arxiv_id":"1803.07635","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-latent","title":"Meta Reinforcement Learning with Latent Variable Gaussian Processes","date":"2018-03-20","arxiv_id":"1803.07551","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-gradient-deep-q-learning","title":"Natural Gradient Deep Q-learning","date":"2018-03-20","arxiv_id":"1803.07482","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-sponsored-search-ranking-strategy","title":"Optimizing Sponsored Search Ranking Strategy by Deep Reinforcement Learning","date":"2018-03-20","arxiv_id":"1803.07347","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduction-for-policy-gradient-with","title":"Variance Reduction for Policy Gradient with Action-Dependent Factorized Baselines","date":"2018-03-20","arxiv_id":"1803.07246","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-doubling-tricks-can-and-cant-do-for","title":"What Doubling Tricks Can and Can't Do for Multi-Armed Bandits","date":"2018-03-19","arxiv_id":"1803.06971","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-neural-architecture-construction-using","title":"Fast Neural Architecture Construction using EnvelopeNets","date":"2018-03-18","arxiv_id":"1803.06744","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-horizon-throughput-maximization-and","title":"Finite Horizon Throughput Maximization and Sensing Optimization in Wireless Powered Devices over Fading Channels","date":"2018-03-17","arxiv_id":"1804.01834","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-visual-question-answering-a-new","title":"Inverse Visual Question Answering: A New Benchmark and VQA Diagnosis Tool","date":"2018-03-16","arxiv_id":"1803.06936","repositories_listed":0,"syntology":null},{"url":null,"slug":"tbd-benchmarking-and-analyzing-deep-neural","title":"TBD: Benchmarking and Analyzing Deep Neural Network Training","date":"2018-03-16","arxiv_id":"1803.06905","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-text-generation-past-present-and","title":"Neural Text Generation: Past, Present and Beyond","date":"2018-03-15","arxiv_id":"1803.07133","repositories_listed":0,"syntology":null},{"url":null,"slug":"rearrangement-with-nonprehensile-manipulation","title":"Rearrangement with Nonprehensile Manipulation Using Deep Reinforcement Learning","date":"2018-03-15","arxiv_id":"1803.05752","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-pedestrian-motion-prediction","title":"Transferable Pedestrian Motion Prediction Models at Intersections","date":"2018-03-15","arxiv_id":"1804.00495","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-speed-and-lane-change-decision","title":"Automated Speed and Lane Change Decision Making using Deep Reinforcement Learning","date":"2018-03-14","arxiv_id":"1803.10056","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-with-concurrent-actions-in","title":"Imitation Learning with Concurrent Actions in 3D Games","date":"2018-03-14","arxiv_id":"1803.05402","repositories_listed":0,"syntology":null},{"url":null,"slug":"measurement-based-adaptation-protocol-with-1","title":"Measurement-based adaptation protocol with quantum reinforcement learning","date":"2018-03-14","arxiv_id":"1803.05340","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-2017-aibirds-competition","title":"The 2017 AIBIRDS Competition","date":"2018-03-14","arxiv_id":"1803.05156","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-reinforcement-learning-with-monte","title":"Active Reinforcement Learning with Monte-Carlo Tree Search","date":"2018-03-13","arxiv_id":"1803.04926","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning","title":"Hierarchical Reinforcement Learning: Approximating Optimal Discounted TSP Using Local Policies","date":"2018-03-13","arxiv_id":"1803.04674","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-with-meta-policy-gradient","title":"Learning to Explore with Meta-Policy Gradient","date":"2018-03-13","arxiv_id":"1803.05044","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-search-in-continuous-action-domains-an","title":"Policy Search in Continuous Action Domains: an Overview","date":"2018-03-13","arxiv_id":"1803.04706","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-nodes-to-networks-evolving-recurrent","title":"From Nodes to Networks: Evolving Recurrent Neural Networks","date":"2018-03-12","arxiv_id":"1803.04439","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-robust-actor-critic-policy-gradient","title":"Soft-Robust Actor-Critic Policy-Gradient","date":"2018-03-11","arxiv_id":"1803.04848","repositories_listed":0,"syntology":null},{"url":null,"slug":"kickstarting-deep-reinforcement-learning","title":"Kickstarting Deep Reinforcement Learning","date":"2018-03-10","arxiv_id":"1803.03835","repositories_listed":0,"syntology":null},{"url":null,"slug":"valuing-knowledge-information-and-agency-in","title":"Valuing knowledge, information and agency in Multi-agent Reinforcement Learning: a case study in smart buildings","date":"2018-03-09","arxiv_id":"1803.03491","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-objective-deep-reinforcement-learning","title":"A Multi-Objective Deep Reinforcement Learning Framework","date":"2018-03-08","arxiv_id":"1803.02965","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepcas-a-deep-reinforcement-learning","title":"DeepCAS: A Deep Reinforcement Learning Algorithm for Control-Aware Scheduling","date":"2018-03-08","arxiv_id":"1803.02998","repositories_listed":0,"syntology":null},{"url":null,"slug":"feudal-reinforcement-learning-for-dialogue","title":"Feudal Reinforcement Learning for Dialogue Management in Large Domains","date":"2018-03-08","arxiv_id":"1803.03232","repositories_listed":0,"syntology":null},{"url":null,"slug":"sa-iga-a-multiagent-reinforcement-learning","title":"SA-IGA: A Multiagent Reinforcement Learning Method Towards Socially Optimal Outcomes","date":"2018-03-08","arxiv_id":"1803.03021","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brandom-ian-view-of-reinforcement-learning","title":"A Brandom-ian view of Reinforcement Learning towards strong-AI","date":"2018-03-07","arxiv_id":"1803.02912","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-action-sequences-from-texts-based","title":"Extracting Action Sequences from Texts Based on Deep Reinforcement Learning","date":"2018-03-07","arxiv_id":"1803.02632","repositories_listed":0,"syntology":null},{"url":null,"slug":"intent-aware-multi-agent-reinforcement","title":"Intent-aware Multi-agent Reinforcement Learning","date":"2018-03-06","arxiv_id":"1803.02018","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-exposure-control-using-adaptive","title":"Personalized Exposure Control Using Adaptive Metering and Reinforcement Learning","date":"2018-03-06","arxiv_id":"1803.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothed-action-value-functions-for-learning","title":"Smoothed Action Value Functions for Learning Gaussian Policies","date":"2018-03-06","arxiv_id":"1803.02348","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-is-more-picking-informative-frames-for","title":"Less Is More: Picking Informative Frames for Video Captioning","date":"2018-03-05","arxiv_id":"1803.01457","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-aware-regret-bounds-for-undiscounted","title":"Variance-Aware Regret Bounds for Undiscounted Reinforcement Learning in MDPs","date":"2018-03-05","arxiv_id":"1803.01626","repositories_listed":0,"syntology":null},{"url":null,"slug":"oil-observational-imitation-learning","title":"OIL: Observational Imitation Learning","date":"2018-03-03","arxiv_id":"1803.01129","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-history-began-from-alexnet-a","title":"The History Began from AlexNet: A Comprehensive Survey on Deep Learning Approaches","date":"2018-03-03","arxiv_id":"1803.01164","repositories_listed":0,"syntology":null}],"record_sha256":"71cff67625dd8a43834decb4c9d50f1e6e56ca8c812652f4831d3bdf6092cf4a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}