{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/39","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":39,"pages_in_order":59,"rows_per_page":100,"rows":[3801,3900],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/38","next":"/task/deep-reinforcement-learning/papers/40","papers":[{"url":null,"slug":"learning-emergent-random-access-protocol-for","title":"Learning Emergent Random Access Protocol for LEO Satellite Networks","date":"2021-12-03","arxiv_id":"2112.01765","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-robustness-of-deep-reinforcement","title":"Adversarial Robustness of Deep Reinforcement Learning based Dynamic Recommender Systems","date":"2021-12-02","arxiv_id":"2112.00973","repositories_listed":0,"syntology":null},{"url":null,"slug":"architecting-and-visualizing-deep","title":"Architecting and Visualizing Deep Reinforcement Learning Models","date":"2021-12-02","arxiv_id":"2112.01451","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-cluster-head-selection-and-trajectory","title":"Joint Cluster Head Selection and Trajectory Planning in UAV-Aided IoT Networks by Reinforcement Learning with Sequential Model","date":"2021-12-01","arxiv_id":"2112.00333","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-tree-interpretation-from-object","title":"Learning Tree Interpretation from Object Representation for Deep Reinforcement Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-arcade-a-configurable-environment-suite","title":"Meta Arcade: A Configurable Environment Suite for Meta-Learning","date":"2021-12-01","arxiv_id":"2112.00583","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-design-for-a-noma-assisted","title":"Energy-Efficient Design for a NOMA assisted STAR-RIS Network with Deep Reinforcement Learning","date":"2021-11-30","arxiv_id":"2111.15464","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-based-implementation-of-colregs-for","title":"Risk-based implementation of COLREGs for autonomous surface vehicles using deep reinforcement learning","date":"2021-11-30","arxiv_id":"2112.00115","repositories_listed":0,"syntology":null},{"url":null,"slug":"saver-safe-learning-based-controller-for-real","title":"SAVER: Safe Learning-Based Controller for Real-Time Voltage Regulation","date":"2021-11-30","arxiv_id":"2111.15152","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepcq-robust-and-scalable-routing-with-multi","title":"DeepCQ+: Robust and Scalable Routing with Multi-Agent Deep Reinforcement Learning for Highly Dynamic Networks","date":"2021-11-29","arxiv_id":"2111.15013","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-can-creativity-occur-in-multi-agent","title":"How Can Creativity Occur in Multi-Agent Systems?","date":"2021-11-29","arxiv_id":"2111.14310","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-model-selection-for-offline-deep-1","title":"Pessimistic Model Selection for Offline Deep Reinforcement Learning","date":"2021-11-29","arxiv_id":"2111.14346","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-the-1","title":"A Reinforcement Learning Approach for the Continuous Electricity Market of Germany: Trading from the Perspective of a Wind Park Operator","date":"2021-11-26","arxiv_id":"2111.13609","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-machine-learning-1","title":"A Comparative Analysis of Machine Learning Techniques for IoT Intrusion Detection","date":"2021-11-25","arxiv_id":"2111.13149","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-skill-adaptation-via-soft-actor-critic","title":"Robot Skill Adaptation via Soft Actor-Critic Gaussian Mixture Models","date":"2021-11-25","arxiv_id":"2111.13129","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-for-deep-reinforcement-learning-in","title":"A Review for Deep Reinforcement Learning in Atari: Benchmarks, Challenges, and Solutions","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/gdi-rethinking-what-makes-reinforcement-1","slug":"gdi-rethinking-what-makes-reinforcement-1","title":"GDI: Rethinking What Makes Reinforcement Learning Different from Supervised Learning","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-multi-goal-exploration","title":"Adaptive Multi-Goal Exploration","date":"2021-11-23","arxiv_id":"2111.12045","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-gpu-compiler-heuristics-using","title":"Generating GPU Compiler Heuristics using Reinforcement Learning","date":"2021-11-23","arxiv_id":"2111.12055","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-collaborative-deep","title":"Semantic-Aware Collaborative Deep Reinforcement Learning Over Wireless Cellular Networks","date":"2021-11-23","arxiv_id":"2111.12064","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-bayesian-deep-reinforcement","title":"Multi-agent Bayesian Deep Reinforcement Learning for Microgrid Energy Management under Communication Failures","date":"2021-11-22","arxiv_id":"2111.11868","repositories_listed":0,"syntology":null},{"url":null,"slug":"renewable-energy-integration-and-microgrid","title":"Renewable energy integration and microgrid energy trading using multi-agent deep reinforcement learning","date":"2021-11-21","arxiv_id":"2111.10898","repositories_listed":0,"syntology":null},{"url":null,"slug":"vulcan-solving-the-steiner-tree-problem-with","title":"Vulcan: Solving the Steiner Tree Problem with Graph Neural Networks and Deep Reinforcement Learning","date":"2021-11-21","arxiv_id":"2111.10810","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-low-rank-q-matrix","title":"Uncertainty-aware Low-Rank Q-Matrix Estimation for Deep Reinforcement Learning","date":"2021-11-19","arxiv_id":"2111.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-generalisation-in-deep","title":"A Survey of Zero-shot Generalisation in Deep Reinforcement Learning","date":"2021-11-18","arxiv_id":"2111.09794","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggressive-q-learning-with-ensembles-1","title":"Aggressive Q-Learning with Ensembles: Achieving Both High Sample Efficiency and High Asymptotic Performance","date":"2021-11-17","arxiv_id":"2111.09159","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-entity","title":"Deep Reinforcement Learning for Entity Alignment","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/gri-general-reinforced-imitation-and-its","slug":"gri-general-reinforced-imitation-and-its","title":"GRI: General Reinforced Imitation and its Application to Vision-Based Autonomous Driving","date":"2021-11-16","arxiv_id":"2111.08575","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-shallow","title":"Deep Reinforcement Learning with Shallow Controllers: An Experimental Application to PID Tuning","date":"2021-11-13","arxiv_id":"2111.07171","repositories_listed":0,"syntology":null},{"url":null,"slug":"obstacle-avoidance-for-uas-in-continuous","title":"Obstacle Avoidance for UAS in Continuous Action Space Using Deep Reinforcement Learning","date":"2021-11-13","arxiv_id":"2111.07037","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deep-reinforcement-learning-for-2","title":"Robust Deep Reinforcement Learning for Extractive Legal Summarization","date":"2021-11-13","arxiv_id":"2111.07158","repositories_listed":0,"syntology":null},{"url":null,"slug":"awd3-dynamic-reduction-of-the-estimation-bias","title":"AWD3: Dynamic Reduction of the Estimation Bias","date":"2021-11-12","arxiv_id":"2111.06780","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-human-level-driving-in-gran-turismo","title":"Expert Human-Level Driving in Gran Turismo Sport Using Deep Reinforcement Learning with Image-based Representation","date":"2021-11-11","arxiv_id":"2111.06449","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-before-you-leap-safe-model-based","title":"Look Before You Leap: Safe Model-Based Reinforcement Learning with Human Intervention","date":"2021-11-10","arxiv_id":"2111.05819","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-deep-reinforcement-learning-for-1","title":"Explainable Deep Reinforcement Learning for Portfolio Management: An Empirical Approach","date":"2021-11-07","arxiv_id":"2111.03995","repositories_listed":0,"syntology":null},{"url":null,"slug":"finrl-deep-reinforcement-learning-framework","title":"FinRL: Deep Reinforcement Learning Framework to Automate Trading in Quantitative Finance","date":"2021-11-07","arxiv_id":"2111.09395","repositories_listed":0,"syntology":null},{"url":null,"slug":"finrl-podracer-high-performance-and-scalable","title":"FinRL-Podracer: High Performance and Scalable Deep Reinforcement Learning for Quantitative Finance","date":"2021-11-07","arxiv_id":"2111.05188","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-10","title":"A Deep Reinforcement Learning Approach for Composing Moving IoT Services","date":"2021-11-06","arxiv_id":"2111.03967","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-collective-behavior-in-newborn","title":"Development of collective behavior in newborn artificial agents","date":"2021-11-06","arxiv_id":"2111.03796","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rna-secondary-structure-design","title":"Improving RNA Secondary Structure Design using Deep Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.04504","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacking-deep-reinforcement-learning-based","title":"Attacking Deep Reinforcement Learning-Based Traffic Signal Control Systems with Colluding Vehicles","date":"2021-11-04","arxiv_id":"2111.02845","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-versus-marginal-shapley-values-for","title":"Causal versus Marginal Shapley Values for Robotic Lever Manipulation Controlled using Deep Reinforcement Learning","date":"2021-11-04","arxiv_id":"2111.02936","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-a-fly-mimicking-flyer-in-complex","title":"Control of a fly-mimicking flyer in complex flow using deep reinforcement learning","date":"2021-11-04","arxiv_id":"2111.03454","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-understanding-of-default-policies","title":"Towards an Understanding of Default Policies in Multitask Policy Optimization","date":"2021-11-04","arxiv_id":"2111.02994","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-attack-mitigation-for-industrial","title":"Autonomous Attack Mitigation for Industrial Control Systems","date":"2021-11-03","arxiv_id":"2111.02445","repositories_listed":0,"syntology":null},{"url":null,"slug":"deployment-optimization-for-shared-e-mobility","title":"Deployment Optimization for Shared e-Mobility Systems with Multi-agent Deep Neural Search","date":"2021-11-03","arxiv_id":"2111.02149","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-8","title":"Multi-Agent Deep Reinforcement Learning For Optimising Energy Efficiency of Fixed-Wing UAV Cellular Access Points","date":"2021-11-03","arxiv_id":"2111.02258","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-service-provisioning-in-nfv-enabled","title":"Online Service Provisioning in NFV-enabled Networks Using Deep Reinforcement Learning","date":"2021-11-03","arxiv_id":"2111.02209","repositories_listed":0,"syntology":null},{"url":null,"slug":"weighted-quantum-channel-compiling-through","title":"Weighted Quantum Channel Compiling through Proximal Policy Optimization","date":"2021-11-03","arxiv_id":"2111.02426","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-robot-do-i-need-fast-co-adaptation-of","title":"What Robot do I Need? Fast Co-Adaptation of Morphology and Control using Graph Neural Networks","date":"2021-11-03","arxiv_id":"2111.02371","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedgraph-federated-graph-learning-with","title":"FedGraph: Federated Graph Learning with Intelligent Sampling","date":"2021-11-02","arxiv_id":"2111.01370","repositories_listed":0,"syntology":null},{"url":null,"slug":"onslicing-online-end-to-end-network-slicing","title":"OnSlicing: Online End-to-End Network Slicing with Reinforcement Learning","date":"2021-11-02","arxiv_id":"2111.01616","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-dialogue-complementary-policy","title":"Efficient Dialogue Complementary Policy Learning via Deep Q-network Policy and Episodic Memory Policy","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-aided-crop-yield","title":"Machine Learning aided Crop Yield Optimization","date":"2021-11-01","arxiv_id":"2111.00963","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewards-with-negative-examples-for-reinforced","title":"Rewards with Negative Examples for Reinforced Topic-Focused Abstractive Summarization","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"brick-by-brick-combinatorial-construction","title":"Brick-by-Brick: Combinatorial Construction with Deep Reinforcement Learning","date":"2021-10-29","arxiv_id":"2110.15481","repositories_listed":0,"syntology":null},{"url":null,"slug":"def-drel-systematic-deployment-of-serverless","title":"DeF-DReL: Systematic Deployment of Serverless Functions in Fog and Cloud environments using Deep Reinforcement Learning","date":"2021-10-29","arxiv_id":"2110.15702","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-sample-efficient-deep-reinforcement","title":"A Novel Sample-efficient Deep Reinforcement Learning with Episodic Policy Transfer for PID-Based Control in Cardiac Catheterization Robots","date":"2021-10-28","arxiv_id":"2110.14941","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-deep-q-learning-framework-for","title":"Cooperative Deep $Q$-learning Framework for Environments Providing Image Feedback","date":"2021-10-28","arxiv_id":"2110.15305","repositories_listed":0,"syntology":null},{"url":null,"slug":"d2rlir-an-improved-and-diversified-ranking","title":"D2RLIR : an improved and diversified ranking function in interactive recommendation systems based on deep reinforcement learning","date":"2021-10-28","arxiv_id":"2110.15089","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-aided-packet","title":"Deep Reinforcement Learning Aided Packet-Routing For Aeronautical Ad-Hoc Networks Formed by Passenger Planes","date":"2021-10-28","arxiv_id":"2110.15146","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-heuristics-constraint-optimization","title":"Comparing Heuristics, Constraint Optimization, and Reinforcement Learning for an Industrial 2D Packing Problem","date":"2021-10-27","arxiv_id":"2110.14535","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-diverse-policies-in-moba-games-via","title":"Learning Diverse Policies in MOBA Games via Macro-Goals","date":"2021-10-27","arxiv_id":"2110.14221","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-demonstrations-with-sacr2-soft","title":"Learning from demonstrations with SACR2: Soft Actor-Critic with Reward Relabeling","date":"2021-10-27","arxiv_id":"2110.14464","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dpdk-based-acceleration-method-for","title":"Accelerating Distributed Deep Reinforcement Learning by In-Network Experience Sampling","date":"2021-10-26","arxiv_id":"2110.13506","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-multi-agent-reinforcement","title":"Applications of Multi-Agent Reinforcement Learning in Future Internet: A Comprehensive Survey","date":"2021-10-26","arxiv_id":"2110.13484","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-multi-agent-deep-reinforcement","title":"Distributed Multi-Agent Deep Reinforcement Learning Framework for Whole-building HVAC Control","date":"2021-10-26","arxiv_id":"2110.13450","repositories_listed":0,"syntology":null},{"url":null,"slug":"hinge-policy-optimization-rethinking-policy-1","title":"Neural PPO-Clip Attains Global Optimality: A Hinge Loss Perspective","date":"2021-10-26","arxiv_id":"2110.13799","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-9","title":"A Deep Reinforcement Learning Approach for Audio-based Navigation and Audio Source Localization in Multi-speaker Environments","date":"2021-10-25","arxiv_id":"2110.12778","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributed-deep-reinforcement-learning","title":"A Distributed Deep Reinforcement Learning Technique for Application Placement in Edge and Fog Computing Environments","date":"2021-10-24","arxiv_id":"2110.12415","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-simultaneous","title":"Deep Reinforcement Learning for Simultaneous Sensing and Channel Access in Cognitive Networks","date":"2021-10-24","arxiv_id":"2110.14541","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-distributed-actor-critic-architecture","title":"Fully Distributed Actor-Critic Architecture for Multitask Deep Reinforcement Learning","date":"2021-10-23","arxiv_id":"2110.12306","repositories_listed":0,"syntology":null},{"url":null,"slug":"anti-concentrated-confidence-bonuses-for-1","title":"Anti-Concentrated Confidence Bonuses for Scalable Exploration","date":"2021-10-21","arxiv_id":"2110.11202","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-generative-models-in-engineering-design","title":"Deep Generative Models in Engineering Design: A Review","date":"2021-10-21","arxiv_id":"2110.10863","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-online-2","title":"Deep Reinforcement Learning for Online Control of Stochastic Partial Differential Equations","date":"2021-10-21","arxiv_id":"2110.11265","repositories_listed":0,"syntology":null},{"url":null,"slug":"locality-sensitive-experience-replay-for","title":"Locality-Sensitive Experience Replay for Online Recommendation","date":"2021-10-21","arxiv_id":"2110.10850","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuro-symbolic-reinforcement-learning-with","title":"Neuro-Symbolic Reinforcement Learning with First-Order Logic","date":"2021-10-21","arxiv_id":"2110.10963","repositories_listed":0,"syntology":null},{"url":null,"slug":"cim-ppo-proximal-policy-optimization-with-liu","title":"CIM-PPO:Proximal Policy Optimization with Liu-Correntropy Induced Metric","date":"2021-10-20","arxiv_id":"2110.10522","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-reinforcement-learning-for-dc-dc","title":"Transferring Reinforcement Learning for DC-DC Buck Converter Control via Duty Ratio Mapping: From Simulation to Implementation","date":"2021-10-20","arxiv_id":"2110.10490","repositories_listed":0,"syntology":null},{"url":null,"slug":"aesthetic-photo-collage-with-deep","title":"Aesthetic Photo Collage with Deep Reinforcement Learning","date":"2021-10-19","arxiv_id":"2110.09775","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedparking-a-federated-learning-based-parking","title":"FedParking: A Federated Learning based Parking Space Estimation with Parked Vehicle assisted Edge Computing","date":"2021-10-19","arxiv_id":"2110.12876","repositories_listed":0,"syntology":null},{"url":null,"slug":"embracing-advanced-ai-ml-to-help-investors","title":"Embracing advanced AI/ML to help investors achieve success: Vanguard Reinforcement Learning for Financial Goal Planning","date":"2021-10-18","arxiv_id":"2110.12003","repositories_listed":0,"syntology":null},{"url":null,"slug":"damped-anderson-mixing-for-deep-reinforcement","title":"Damped Anderson Mixing for Deep Reinforcement Learning: Acceleration, Convergence, and Stabilization","date":"2021-10-17","arxiv_id":"2110.08896","repositories_listed":0,"syntology":null},{"url":null,"slug":"case-based-reasoning-for-better-1","title":"Case-based Reasoning for Better Generalization in Textual Reinforcement Learning","date":"2021-10-16","arxiv_id":"2110.08470","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotion-style-transfer-with-a-specified","title":"Emotion Style Transfer with a Specified Intensity Using Deep Reinforcement Learning","date":"2021-10-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lifting-the-veil-on-hyper-parameters-for","title":"Lifting the veil on hyper-parameters for value-based deep reinforcement learning","date":"2021-10-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"local-advantage-actor-critic-for-robust-multi","title":"Local Advantage Actor-Critic for Robust Multi-Agent Deep Reinforcement Learning","date":"2021-10-16","arxiv_id":"2110.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-broad-persistent-advising-approach-for-deep","title":"A Broad-persistent Advising Approach for Deep Interactive Reinforcement Learning in Robotic Environments","date":"2021-10-15","arxiv_id":"2110.08003","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-in-urban-environments-amongst","title":"Navigation In Urban Environments Amongst Pedestrians Using Multi-Objective Deep Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05205","repositories_listed":0,"syntology":null},{"url":null,"slug":"rein-2-giving-birth-to-prepared-reinforcement","title":"REIN-2: Giving Birth to Prepared Reinforcement Learning Agents Using Reinforcement Learning Agents","date":"2021-10-11","arxiv_id":"2110.05128","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-traffic-signal-controls-using-fog","title":"Scalable Traffic Signal Controls using Fog-Cloud Based Multiagent Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05564","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-optimizing","title":"Deep Reinforcement Learning for Optimizing RIS-Assisted HD-FD Wireless Systems","date":"2021-10-10","arxiv_id":"2110.04859","repositories_listed":0,"syntology":null},{"url":null,"slug":"hard-instance-learning-for-quantum-adiabatic","title":"Hard instance learning for quantum adiabatic prime factorization","date":"2021-10-10","arxiv_id":"2110.04782","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-condition-multi-objective-optimization","title":"Multi-condition multi-objective optimization using deep reinforcement learning","date":"2021-10-10","arxiv_id":"2110.05945","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretically-principled-deep-rl-acceleration","title":"Theoretically Principled Deep RL Acceleration via Nearest Neighbor Function Approximation","date":"2021-10-09","arxiv_id":"2110.04422","repositories_listed":0,"syntology":null},{"url":null,"slug":"cheerbots-chatbots-toward-empathy-and","title":"CheerBots: Chatbots toward Empathy and Emotionusing Reinforcement Learning","date":"2021-10-08","arxiv_id":"2110.03949","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-deep-reinforcement-learning-agents","title":"Explaining Deep Reinforcement Learning Agents In The Atari Domain through a Surrogate Model","date":"2021-10-07","arxiv_id":"2110.03184","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-in-deep-rl-for-tsp-problems","title":"Generalization in Deep RL for TSP Problems via Equivariance and Local Search","date":"2021-10-07","arxiv_id":"2110.03595","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-pessimism-for-robust-and-efficient","title":"Learning Pessimism for Robust and Efficient Off-Policy Reinforcement Learning","date":"2021-10-07","arxiv_id":"2110.03375","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-lever-manipulation-using-hindsight","title":"Robotic Lever Manipulation using Hindsight Experience Replay and Shapley Additive Explanations","date":"2021-10-07","arxiv_id":"2110.03292","repositories_listed":0,"syntology":null}],"record_sha256":"42308bceec2f36f4ef7e305e4644a2d52e782d3be03025c8a066b51597e7f803","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}