{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/116","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":116,"pages_in_order":132,"rows_per_page":100,"rows":[11501,11600],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/115","next":"/task/reinforcement-learning/papers/117","papers":[{"url":null,"slug":"reward-estimation-variance-elimination-in","title":"Reward-estimation variance elimination in sequential decision processes","date":"2018-11-15","arxiv_id":"1811.06225","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-utility-of-sparse-representations-for","title":"The Utility of Sparse Representations for Control in Reinforcement Learning","date":"2018-11-15","arxiv_id":"1811.06626","repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-bayesian-ambiguity-sets-for-robust-mdps","title":"Tight Bayesian Ambiguity Sets for Robust MDPs","date":"2018-11-15","arxiv_id":"1811.06512","repositories_listed":0,"syntology":null},{"url":null,"slug":"woulda-coulda-shoulda-counterfactually-guided","title":"Woulda, Coulda, Shoulda: Counterfactually-Guided Policy Search","date":"2018-11-15","arxiv_id":"1811.06272","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-reinforcement-learning-in-factored","title":"Bayesian Reinforcement Learning in Factored POMDPs","date":"2018-11-14","arxiv_id":"1811.05612","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-addictive-behaviors-in","title":"Emergence of Addictive Behaviors in Reinforcement Learning Agents","date":"2018-11-14","arxiv_id":"1811.05590","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-multiple-diverse-responses-for","title":"Generating Multiple Diverse Responses for Short-Text Conversation","date":"2018-11-14","arxiv_id":"1811.05696","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-interactive-recommendation-with","title":"Large-scale Interactive Recommendation with Tree-structured Policy Gradient","date":"2018-11-14","arxiv_id":"1811.05869","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-captioning-based-on-a-hierarchical","title":"Image Captioning Based on a Hierarchical Attention Mechanism and Policy Gradient Optimization","date":"2018-11-13","arxiv_id":"1811.05253","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-the-dynamic-joint-policy-of","title":"Modelling the Dynamic Joint Policy of Teammates with Attention Multi-agent DDPG","date":"2018-11-13","arxiv_id":"1811.07029","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinating-disaster-emergency-response-with","title":"Coordinating Disaster Emergency Response with Heuristic Reinforcement Learning","date":"2018-11-12","arxiv_id":"1811.05010","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-weighted-evolution-strategies","title":"Importance Weighted Evolution Strategies","date":"2018-11-12","arxiv_id":"1811.04624","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-data-augmentation-policies-using","title":"Learning data augmentation policies using augmented random search","date":"2018-11-12","arxiv_id":"1811.04768","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-temporal-point-processes-via","title":"Learning Temporal Point Processes via Reinforcement Learning","date":"2018-11-12","arxiv_id":"1811.05016","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-assistance-system-for-quadcopter","title":"Navigating Assistance System for Quadcopter with Deep Reinforcement Learning","date":"2018-11-12","arxiv_id":"1811.04584","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-initial-attempt-of-combining-visual","title":"An initial attempt of combining visual selective attention with deep reinforcement learning","date":"2018-11-11","arxiv_id":"1811.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimal-control-view-of-adversarial","title":"An Optimal Control View of Adversarial Machine Learning","date":"2018-11-11","arxiv_id":"1811.04422","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-taxi-carpool-policies-via","title":"Optimizing Taxi Carpool Policies via Reinforcement Learning and Spatio-Temporal Mining","date":"2018-11-11","arxiv_id":"1811.04345","repositories_listed":0,"syntology":null},{"url":null,"slug":"product-title-refinement-via-multi-modal","title":"Product Title Refinement via Multi-Modal Generative Adversarial Learning","date":"2018-11-11","arxiv_id":"1811.04498","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-governing-agents-efficacy-action","title":"Towards Governing Agent's Efficacy: Action-Conditional $β$-VAE for Deep Transparent Reinforcement Learning","date":"2018-11-11","arxiv_id":"1811.04350","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-shaping-strategies-in-human-in-the","title":"Learning Shaping Strategies in Human-in-the-loop Interactive Reinforcement Learning","date":"2018-11-10","arxiv_id":"1811.04272","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-speech","title":"Reinforcement Learning Based Speech Enhancement for Robust Speech Recognition","date":"2018-11-10","arxiv_id":"1811.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-policy-learning-based-on","title":"Sample-Efficient Policy Learning based on Completely Behavior Cloning","date":"2018-11-09","arxiv_id":"1811.03853","repositories_listed":0,"syntology":null},{"url":null,"slug":"correlation-filter-selection-for-visual","title":"Correlation Filter Selection for Visual Tracking Using Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03196","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-multi-objective","title":"Meta-Learning for Multi-objective Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03376","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-architecture-for-starcraft-ii-with","title":"Modular Architecture for StarCraft II with Deep Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03555","repositories_listed":0,"syntology":null},{"url":null,"slug":"baselines-for-reinforcement-learning-in-text","title":"Baselines for Reinforcement Learning in Text Games","date":"2018-11-07","arxiv_id":"1811.02872","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-certificates-towards-accountable","title":"Policy Certificates: Towards Accountable Reinforcement Learning","date":"2018-11-07","arxiv_id":"1811.03056","repositories_listed":0,"syntology":null},{"url":null,"slug":"roboturk-a-crowdsourcing-platform-for-robotic","title":"RoboTurk: A Crowdsourcing Platform for Robotic Skill Learning through Imitation","date":"2018-11-07","arxiv_id":"1811.02790","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-finding-failure","title":"Adaptive Stress Testing: Finding Likely Failure Events with Reinforcement Learning","date":"2018-11-06","arxiv_id":"1811.02188","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-green","title":"Deep Reinforcement Learning for Green Security Games with Real-Time Information","date":"2018-11-06","arxiv_id":"1811.02483","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-optimization-in-deep-q-learning","title":"Deep Reinforcement Learning via L-BFGS Optimization","date":"2018-11-06","arxiv_id":"1811.02693","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-continual-learning-in-medical-imaging","title":"Towards continual learning in medical imaging","date":"2018-11-06","arxiv_id":"1811.02496","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-subgoal-graphs-with-reinforcement","title":"Combining Subgoal Graphs with Reinforcement Learning to Build a Rational Pathfinder","date":"2018-11-05","arxiv_id":"1811.01700","repositories_listed":0,"syntology":null},{"url":"/paper/contingency-aware-exploration-in","slug":"contingency-aware-exploration-in","title":"Contingency-Aware Exploration in Reinforcement Learning","date":"2018-11-05","arxiv_id":"1811.01483","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-engineering-systems-with-large-state","title":"Managing engineering systems with large state and action spaces through deep reinforcement learning","date":"2018-11-05","arxiv_id":"1811.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-dynamic-model","title":"Reinforcement Learning based Dynamic Model Selection for Short-Term Load Forecasting","date":"2018-11-05","arxiv_id":"1811.01846","repositories_listed":0,"syntology":null},{"url":null,"slug":"releq-an-automatic-reinforcement-learning","title":"ReLeQ: A Reinforcement Learning Approach for Deep Quantization of Neural Networks","date":"2018-11-05","arxiv_id":"1811.01704","repositories_listed":0,"syntology":null},{"url":null,"slug":"relation-mention-extraction-from-noisy-data","title":"Relation Mention Extraction from Noisy Data with Hierarchical Reinforcement Learning","date":"2018-11-03","arxiv_id":"1811.01237","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-theorem-proving-in-intuitionistic","title":"Automated Theorem Proving in Intuitionistic Propositional Logic by Deep Reinforcement Learning","date":"2018-11-02","arxiv_id":"1811.00796","repositories_listed":0,"syntology":null},{"url":null,"slug":"dantzig-selector-with-an-approximately","title":"Dantzig Selector with an Approximately Optimal Denoising Matrix and its Application to Reinforcement Learning","date":"2018-11-02","arxiv_id":"1811.00958","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-generation-with-guider-network","title":"Sequence Generation with Guider Network","date":"2018-11-02","arxiv_id":"1811.00696","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-dynamic-oracle-for-dependency","title":"Approximate Dynamic Oracle for Dependency Parsing with Reinforcement Learning","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-modeling-for-query-expansion-and","title":"Joint Modeling for Query Expansion and Information Extraction with Reinforcement Learning","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"macquarie-university-at-bioasq-6b-deep","title":"Macquarie University at BioASQ 6b: Deep learning and deep reinforcement learning for query-based summarisation","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-a-social-robots-humor-with-natural","title":"Shaping a social robot's humor with Natural Language Generation and socially-aware reinforcement learning","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sdrl-interpretable-and-data-efficient-deep","title":"SDRL: Interpretable and Data-efficient Deep Reinforcement Learning Leveraging Symbolic Planning","date":"2018-10-31","arxiv_id":"1811.00090","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-learning-of-deep-neural-networks","title":"Structure Learning of Deep Neural Networks with Q-Learning","date":"2018-10-31","arxiv_id":"1810.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-simple-approach-to-multi-step-model","title":"Towards a Simple Approach to Multi-step Model-based Reinforcement Learning","date":"2018-10-31","arxiv_id":"1811.00128","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-importance-sampling-for-off-policy","title":"Relative Importance Sampling for off-Policy Actor-Critic in Deep Reinforcement Learning","date":"2018-10-30","arxiv_id":"1810.12558","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-vehicle-swarms-a-novel-perspective-on","title":"Social Vehicle Swarms: A Novel Perspective on Social-aware Vehicular Communication Architecture","date":"2018-10-29","arxiv_id":"1810.11947","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributive-dynamic-spectrum-access-through","title":"Distributive Dynamic Spectrum Access through Deep Reinforcement Learning: A Reservoir Computing Based Approach","date":"2018-10-28","arxiv_id":"1810.11758","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-abstract-options","title":"Learning Abstract Options","date":"2018-10-27","arxiv_id":"1810.11583","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-evaluation-of-contextual-policy","title":"Empirical Evaluation of Contextual Policy Search with a Comparison-based Surrogate Model and Active Covariance Matrix Adaptation","date":"2018-10-26","arxiv_id":"1810.11491","repositories_listed":0,"syntology":null},{"url":null,"slug":"stability-certified-reinforcement-learning-a","title":"Stability-certified reinforcement learning: A control-theoretic perspective","date":"2018-10-26","arxiv_id":"1810.11505","repositories_listed":0,"syntology":null},{"url":null,"slug":"tarmac-targeted-multi-agent-communication","title":"TarMAC: Targeted Multi-Agent Communication","date":"2018-10-26","arxiv_id":"1810.11187","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-variable-speed-limits-control","title":"Differential Variable Speed Limits Control for Freeway Recurrent Bottlenecks via Deep Reinforcement learning","date":"2018-10-25","arxiv_id":"1810.10952","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-modeling-game-for-deriving-theoretical","title":"Meta-modeling game for deriving theoretical-consistent, micro-structural-based traction-separation laws via deep reinforcement learning","date":"2018-10-24","arxiv_id":"1810.10535","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-based-1","title":"Multi-Agent Reinforcement Learning Based Resource Allocation for UAV Networks","date":"2018-10-24","arxiv_id":"1810.10408","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-learning-of-nonprehensile","title":"Sample-Efficient Learning of Nonprehensile Manipulation Policies via Physics-Based Informed State Distributions","date":"2018-10-24","arxiv_id":"1810.10654","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-approaches-for-reinforcement","title":"Hierarchical Approaches for Reinforcement Learning in Parameterized Action Space","date":"2018-10-23","arxiv_id":"1810.09656","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-first-to-spike-policies-for","title":"Learning First-to-Spike Policies for Neuromorphic Control Using Policy Gradients","date":"2018-10-23","arxiv_id":"1810.09977","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-representations-in-model-free","title":"Learning Representations in Model-Free Hierarchical Reinforcement Learning","date":"2018-10-23","arxiv_id":"1810.10096","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-faults-in-our-pi-stars-security-issues","title":"The Faults in Our Pi Stars: Security Issues and Open Challenges in Deep Reinforcement Learning","date":"2018-10-23","arxiv_id":"1810.10369","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-actor-critic-with-generative","title":"Multi-Agent Actor-Critic with Generative Cooperative Policy Network","date":"2018-10-22","arxiv_id":"1810.09206","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning-a","title":"Risk-Sensitive Reinforcement Learning via Policy Gradient Search","date":"2018-10-22","arxiv_id":"1810.09126","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-inverse-reinforcement-learners-via","title":"Teaching Inverse Reinforcement Learners via Features and Demonstrations","date":"2018-10-21","arxiv_id":"1810.08926","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-self-explanation-of-behavior-for","title":"Autonomous Self-Explanation of Behavior for Interactive Reinforcement Learning Agents","date":"2018-10-20","arxiv_id":"1810.08811","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-model","title":"Safe Reinforcement Learning with Model Uncertainty Estimates","date":"2018-10-19","arxiv_id":"1810.08700","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-versus-multi-agent-learning","title":"Transfer Learning versus Multi-agent Learning regarding Distributed Decision-Making in Highway Traffic","date":"2018-10-19","arxiv_id":"1810.08515","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-deep-reinforcement-learning","title":"Applications of Deep Reinforcement Learning in Communications and Networking: A Survey","date":"2018-10-18","arxiv_id":"1810.07862","repositories_listed":0,"syntology":null},{"url":null,"slug":"trust-region-policy-optimization-for-pomdps","title":"Policy Gradient in Partially Observable Environments: Approximation and Convergence","date":"2018-10-18","arxiv_id":"1810.07900","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-observation-learning","title":"O2A: One-shot Observational learning with Action vectors","date":"2018-10-17","arxiv_id":"1810.07483","repositories_listed":0,"syntology":null},{"url":null,"slug":"at-human-speed-deep-reinforcement-learning","title":"At Human Speed: Deep Reinforcement Learning with Action Delay","date":"2018-10-16","arxiv_id":"1810.07286","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-learning-abstract-discrete","title":"Incremental learning abstract discrete planning domains and mappings to continuous perceptions","date":"2018-10-16","arxiv_id":"1810.07096","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-concept-of-criticality-in-reinforcement","title":"The Concept of Criticality in Reinforcement Learning","date":"2018-10-16","arxiv_id":"1810.07254","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-agent-behavior-over-long-time","title":"Optimizing Agent Behavior over Long Time Scales by Transporting Value","date":"2018-10-15","arxiv_id":"1810.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-for-the","title":"Using Deep Reinforcement Learning for the Continuous Control of Robotic Arms","date":"2018-10-15","arxiv_id":"1810.06746","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexterous-manipulation-with-deep","title":"Dexterous Manipulation with Deep Reinforcement Learning: Efficient, General, and Low-Cost","date":"2018-10-14","arxiv_id":"1810.06045","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-can-play-that-game-an-adversarial","title":"Two Can Play That Game: An Adversarial Evaluation of a Cyber-alert Inspection System","date":"2018-10-13","arxiv_id":"1810.05921","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-inference-of-self-intention","title":"Bayesian Inference of Self-intention Attributed by Observer","date":"2018-10-12","arxiv_id":"1810.05564","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-multiagent-deep-reinforcement-learning-the","title":"A Survey and Critique of Multiagent Deep Reinforcement Learning","date":"2018-10-12","arxiv_id":"1810.05587","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-hierarchical-learning-path-design","title":"Optimal Hierarchical Learning Path Design with Reinforcement Learning","date":"2018-10-12","arxiv_id":"1810.05347","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-learning-of-movement-prediction-in","title":"Sequential Learning of Movement Prediction in Dynamic Environments using LSTM Autoencoder","date":"2018-10-12","arxiv_id":"1810.05394","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-text-generation-without","title":"Adversarial Text Generation Without Reinforcement Learning","date":"2018-10-11","arxiv_id":"1810.06640","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-design-for-active-sequential","title":"Policy Design for Active Sequential Hypothesis Testing using Deep Learning","date":"2018-10-11","arxiv_id":"1810.04859","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-laplacian-in-rl-learning-representations","title":"The Laplacian in RL: Learning Representations with Efficient Approximations","date":"2018-10-10","arxiv_id":"1810.04586","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributed-reinforcement-learning-solution","title":"A Distributed Reinforcement Learning Solution With Knowledge Transfer Capability for A Bike Rebalancing Problem","date":"2018-10-09","arxiv_id":"1810.04058","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-state-representation-learning-for","title":"Continual State Representation Learning for Reinforcement Learning using Generative Replay","date":"2018-10-09","arxiv_id":"1810.03880","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-wildfire-surveillance-with","title":"Distributed Wildfire Surveillance with Autonomous Aircraft using Deep Reinforcement Learning","date":"2018-10-09","arxiv_id":"1810.04244","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-cognitive-smart-cities-using-big","title":"Enabling Cognitive Smart Cities Using Big Data and Machine Learning: Approaches and Challenges","date":"2018-10-09","arxiv_id":"1810.04107","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-enactive-learning-for","title":"Investigating Enactive Learning for Autonomous Intelligent Agents","date":"2018-10-09","arxiv_id":"1810.04535","repositories_listed":0,"syntology":null},{"url":null,"slug":"realizing-learned-quadruped-locomotion","title":"Realizing Learned Quadruped Locomotion Behaviors through Kinematic Motion Primitives","date":"2018-10-09","arxiv_id":"1810.03842","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for","title":"Actor-Critic Deep Reinforcement Learning for Dynamic Multichannel Access","date":"2018-10-08","arxiv_id":"1810.03695","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for","title":"Multi-agent Deep Reinforcement Learning for Zero Energy Communities","date":"2018-10-08","arxiv_id":"1810.03679","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-evolutionary-learning-method","title":"Reinforcement Evolutionary Learning Method for self-learning","date":"2018-10-07","arxiv_id":"1810.03198","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-simulate","title":"Learning To Simulate","date":"2018-10-05","arxiv_id":"1810.02513","repositories_listed":0,"syntology":null},{"url":null,"slug":"zooming-network","title":"Zooming Network","date":"2018-10-04","arxiv_id":"1810.02114","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-time","title":"Deep Reinforcement Learning for Time Scheduling in RF-Powered Backscatter Cognitive Radio Networks","date":"2018-10-03","arxiv_id":"1810.04520","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-teacher-student-framework-for-maintainable","title":"A Teacher-Student Framework for Maintainable Dialog Manager","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"1aa8c8b46f533f3c436ca16a02886535842e2c76af17fa0d143aae62857e1595","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}