{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ddpg/papers/2","list_of":"/method/ddpg","method":"DDPG","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":218,"counts":{"archive_papers_tagged":218,"with_a_code_link":71,"where_syntology_ran_a_sample":14,"not_listed_spam_title":0,"listed":218,"listed_where_code_ran":14,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":13,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":13,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ddpg","prev":"/method/ddpg","next":"/method/ddpg/papers/3","papers":[{"paper":null,"slug":"asymptotic-convergence-of-deep-multi-agent","title":"3DPG: Distributed Deep Deterministic Policy Gradient Algorithms for Networked Multi-Agent Systems","date":"2022-01-03","arxiv_id":"2201.00570","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-pareto-efficient-fairness-utility","title":"Toward Pareto Efficient Fairness-Utility Trade-off inRecommendation through Reinforcement Learning","date":"2022-01-01","arxiv_id":"2201.00140","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-optimal-power","title":"Deep Reinforcement Learning for Optimal Power Flow with Renewables Using Graph Information","date":"2021-12-22","arxiv_id":"2112.11461","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-control-of-overestimation-bias-for","title":"Automating Control of Overestimation Bias for Reinforcement Learning","date":"2021-10-26","arxiv_id":"2110.13523","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-off-policy-baselines-for-memory","slug":"recurrent-off-policy-baselines-for-memory","title":"Recurrent Off-policy Baselines for Memory-based Continuous Control","date":"2021-10-25","arxiv_id":"2110.12628","n_code_links":1,"syntology":null},{"paper":null,"slug":"computationally-efficient-safe-reinforcement","title":"Computationally Efficient Safe Reinforcement Learning for Power Systems","date":"2021-10-20","arxiv_id":"2110.10333","n_code_links":0,"syntology":null},{"paper":null,"slug":"parallel-actors-and-learners-a-framework-for","title":"Parallel Actors and Learners: A Framework for Generating Scalable RL Implementations","date":"2021-10-03","arxiv_id":"2110.01101","n_code_links":0,"syntology":null},{"paper":null,"slug":"bootstrapped-hindsight-experience-replay-with","title":"Bootstrapped Hindsight Experience replay with Counterintuitive Prioritization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"experience-replay-more-when-it-s-a-key","title":"Experience Replay More When It's a Key Transition in Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-attention-for-off-policy-actor-critic","title":"Meta Attention For Off-Policy Actor-Critic","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"spp-rl-state-planning-policy-reinforcement","title":"SPP-RL: State Planning Policy Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-2","title":"Deep Reinforcement Learning Based Multidimensional Resource Management for Energy Harvesting Cognitive NOMA Communications","date":"2021-09-17","arxiv_id":"2109.09503","n_code_links":0,"syntology":null},{"paper":"/paper/responsive-regulation-of-dynamic-uav","slug":"responsive-regulation-of-dynamic-uav","title":"Responsive Regulation of Dynamic UAV Communication Networks Based on Deep Reinforcement Learning","date":"2021-08-25","arxiv_id":"2108.11012","n_code_links":1,"syntology":null},{"paper":"/paper/safe-deep-reinforcement-learning-for-multi","slug":"safe-deep-reinforcement-learning-for-multi","title":"Safe Deep Reinforcement Learning for Multi-Agent Systems with Continuous Action Spaces","date":"2021-08-09","arxiv_id":"2108.03952","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zisikons/deep-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ai-based-secure-noma-and-cognitive-radio","title":"AI-Based Secure NOMA and Cognitive Radio enabled Green Communications: Channel State Information and Battery Value Uncertainties","date":"2021-06-30","arxiv_id":"2106.15964","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-approach-for-an-irs","title":"A Reinforcement Learning Approach for an IRS-assisted NOMA Network","date":"2021-06-17","arxiv_id":"2106.09611","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-1","title":"Deep Reinforcement Learning Based Optimization for IRS Based UAV-NOMA Downlink Networks","date":"2021-06-17","arxiv_id":"2106.09616","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-continuous-control-with-double","slug":"efficient-continuous-control-with-double","title":"Efficient Continuous Control with Double Actors and Regularized Critics","date":"2021-06-06","arxiv_id":"2106.03050","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-uav","title":"Deep Reinforcement Learning-based UAV Navigation and Control: A Soft Actor-Critic with Hindsight Experience Replay Approach","date":"2021-06-02","arxiv_id":"2106.01016","n_code_links":0,"syntology":null},{"paper":"/paper/improved-exploring-starts-by-kernel-density","slug":"improved-exploring-starts-by-kernel-density","title":"Improved Exploring Starts by Kernel Density Estimation-Based State-Space Coverage Acceleration in Reinforcement Learning","date":"2021-05-19","arxiv_id":"2105.08990","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-deterministic-path-following","title":"Deep Deterministic Path Following","date":"2021-04-13","arxiv_id":"2104.06014","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-controller","title":"Deep Reinforcement Learning Based Controller for Active Heave Compensation","date":"2021-04-12","arxiv_id":"2104.05599","n_code_links":0,"syntology":null},{"paper":null,"slug":"progressive-extension-of-reinforcement","title":"Progressive extension of reinforcement learning action dimension for asymmetric assembly tasks","date":"2021-04-06","arxiv_id":"2104.04078","n_code_links":0,"syntology":null},{"paper":"/paper/self-adaptive-torque-vectoring-controller","slug":"self-adaptive-torque-vectoring-controller","title":"Self-adaptive Torque Vectoring Controller Using Reinforcement Learning","date":"2021-03-27","arxiv_id":"2103.14892","n_code_links":1,"syntology":null},{"paper":null,"slug":"simulation-studies-on-deep-reinforcement","title":"Simulation Studies on Deep Reinforcement Learning for Building Control with Human Interaction","date":"2021-03-14","arxiv_id":"2103.07919","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-driven-mimo-control-of-room-temperature","title":"Data-driven control of room temperature and bidirectional EV charging using deep reinforcement learning: simulations and experiments","date":"2021-03-02","arxiv_id":"2103.01886","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-path-planning-based-on-mpc-and","title":"Multi-Agent Path Planning based on MPC and DDPG","date":"2021-02-26","arxiv_id":"2102.13283","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-car-following-strategy-based-on-deep","title":"Hybrid Car-Following Strategy based on Deep Deterministic Policy Gradient and Cooperative Adaptive Cruise Control","date":"2021-02-24","arxiv_id":"2103.03796","n_code_links":0,"syntology":null},{"paper":null,"slug":"escaping-from-zero-gradient-revisiting-action","title":"Escaping from Zero Gradient: Revisiting Action-Constrained Reinforcement Learning via Frank-Wolfe Policy Optimization","date":"2021-02-22","arxiv_id":"2102.11055","n_code_links":0,"syntology":null},{"paper":"/paper/accelerated-sim-to-real-deep-reinforcement","slug":"accelerated-sim-to-real-deep-reinforcement","title":"Accelerated Sim-to-Real Deep Reinforcement Learning: Learning Collision Avoidance from Human Player","date":"2021-02-21","arxiv_id":"2102.10711","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-symmetric","title":"Deep Reinforcement Learning with Symmetric Prior for Predictive Power Allocation to Mobile Users","date":"2021-02-10","arxiv_id":"2103.13298","n_code_links":0,"syntology":null},{"paper":"/paper/explainable-reinforcement-learning-for","slug":"explainable-reinforcement-learning-for","title":"Explainable Reinforcement Learning for Longitudinal Control","date":"2021-02-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-of-motion-planning-algorithms-for","title":"A review of motion planning algorithms for intelligent robotics","date":"2021-02-04","arxiv_id":"2102.02376","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-for-control-of-valves","slug":"reinforcement-learning-for-control-of-valves","title":"Reinforcement Learning for Control of Valves","date":"2020-12-29","arxiv_id":"2012.14668","n_code_links":2,"syntology":null},{"paper":null,"slug":"2012-11643","title":"myGym: Modular Toolkit for Visuomotor Robotic Tasks","date":"2020-12-21","arxiv_id":"2012.11643","n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-gradient-for-items-recommendation-on","title":"Policy Gradient for items Recommendation on Virtual Taobao","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","n_code_links":1,"syntology":null},{"paper":null,"slug":"virtual-autonomous-driving-with-reinforcement","title":"Virtual Autonomous Driving with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-reservoir-management-through-deep","title":"Efficient Reservoir Management through Deep Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.03822","n_code_links":0,"syntology":null},{"paper":"/paper/finrl-a-deep-reinforcement-learning-library","slug":"finrl-a-deep-reinforcement-learning-library","title":"FinRL: A Deep Reinforcement Learning Library for Automated Stock Trading in Quantitative Finance","date":"2020-11-19","arxiv_id":"2011.09607","n_code_links":6,"syntology":null},{"paper":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-in-electricity","title":"Deep Reinforcement Learning in Electricity Generation Investment for the Minimization of Long-Term Carbon Emissions and Electricity Costs","date":"2020-11-02","arxiv_id":"2011.02342","n_code_links":0,"syntology":null},{"paper":"/paper/self-driving-network-and-service-coordination","slug":"self-driving-network-and-service-coordination","title":"Self-Driving Network and Service Coordination Using Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-coverage-and-capacity-in-cellular","title":"Optimizing Coverage and Capacity in Cellular Networks using Machine Learning","date":"2020-10-22","arxiv_id":"2010.13710","n_code_links":0,"syntology":null},{"paper":null,"slug":"recurrent-distributed-reinforcement-learning","title":"A Learning Approach to Robot-Agnostic Force-Guided High Precision Assembly","date":"2020-10-15","arxiv_id":"2010.08052","n_code_links":0,"syntology":null},{"paper":null,"slug":"hindsight-experience-replay-with-kronecker","title":"Hindsight Experience Replay with Kronecker Product Approximate Curvature","date":"2020-10-09","arxiv_id":"2010.06142","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-assisted-deep-reinforcement","title":"Knowledge-Assisted Deep Reinforcement Learning in 5G Scheduler Design: From Theoretical Framework to Implementation","date":"2020-09-17","arxiv_id":"2009.08346","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimization-driven-hierarchical-learning","title":"Optimization-driven Hierarchical Learning Framework for Wireless Powered Backscatter-aided Relay Communications","date":"2020-08-04","arxiv_id":"2008.01366","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-like-energy-management-based-on-deep","title":"Human-like Energy Management Based on Deep Reinforcement Learning and Historical Driving Experiences","date":"2020-07-16","arxiv_id":"2007.10126","n_code_links":0,"syntology":null},{"paper":null,"slug":"regularly-updated-deterministic-policy","title":"Regularly Updated Deterministic Policy Gradient Algorithm","date":"2020-07-01","arxiv_id":"2007.00169","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributed-uplink-beamforming-in-cell-free","title":"Distributed Uplink Beamforming in Cell-Free Networks Using Deep Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.15138","n_code_links":0,"syntology":null},{"paper":null,"slug":"noise-overestimation-and-exploration-in-deep","title":"Some approaches used to overcome overestimation in Deep Reinforcement Learning algorithms","date":"2020-06-25","arxiv_id":"2006.14167","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-effect-of-multi-step-methods-on","title":"The Effect of Multi-step Methods on Overestimation in Deep Reinforcement Learning","date":"2020-06-23","arxiv_id":"2006.12692","n_code_links":0,"syntology":null},{"paper":null,"slug":"reducing-estimation-bias-via-weighted-delayed","title":"WD3: Taming the Estimation Bias in Deep Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.12622","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-online-evolving-framework-for-advancing","title":"An online evolving framework for advancing reinforcement-learning based automated vehicle control","date":"2020-06-15","arxiv_id":"2006.08092","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimization-driven-deep-reinforcement","title":"Optimization-driven Deep Reinforcement Learning for Robust Beamforming in IRS-assisted Wireless Communications","date":"2020-05-25","arxiv_id":"2005.11885","n_code_links":0,"syntology":null},{"paper":null,"slug":"pbcs-efficient-exploration-and-exploitation","title":"PBCS : Efficient Exploration and Exploitation Using a Synergy between Reinforcement Learning and Motion Planning","date":"2020-04-24","arxiv_id":"2004.11667","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-based-actor-critic-gan-drl-actor-critic","title":"Model-based actor-critic: GAN (model generator) + DRL (actor-critic) => AGI","date":"2020-04-04","arxiv_id":"2004.04574","n_code_links":0,"syntology":null},{"paper":"/paper/obstacle-avoidance-and-navigation-utilizing","slug":"obstacle-avoidance-and-navigation-utilizing","title":"Obstacle Avoidance and Navigation Utilizing Reinforcement Learning with Reward Shaping","date":"2020-03-28","arxiv_id":"2003.12863","n_code_links":1,"syntology":null},{"paper":null,"slug":"accelerating-deep-reinforcement-learning-with","title":"Accelerating Deep Reinforcement Learning With the Aid of Partial Model: Energy-Efficient Predictive Video Streaming","date":"2020-03-21","arxiv_id":"2003.09708","n_code_links":0,"syntology":null},{"paper":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","n_code_links":4,"syntology":null},{"paper":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","n_code_links":1,"syntology":null},{"paper":"/paper/online-meta-critic-learning-for-off-policy-1","slug":"online-meta-critic-learning-for-off-policy-1","title":"Online Meta-Critic Learning for Off-Policy Actor-Critic Methods","date":"2020-03-11","arxiv_id":"2003.05334","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","official":{"repos":["zwfightzw/Meta-Critic"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamic-experience-replay","title":"Dynamic Experience Replay","date":"2020-03-04","arxiv_id":"2003.02372","n_code_links":0,"syntology":null},{"paper":"/paper/contention-window-optimization-in-ieee","slug":"contention-window-optimization-in-ieee","title":"Contention Window Optimization in IEEE 802.11ax Networks with Deep Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01492","n_code_links":1,"syntology":null},{"paper":"/paper/reinforcement-co-learning-of-deep-and-spiking","slug":"reinforcement-co-learning-of-deep-and-spiking","title":"Reinforcement co-Learning of Deep and Spiking Neural Networks for Energy-Efficient Mapless Navigation with Neuromorphic Hardware","date":"2020-03-02","arxiv_id":"2003.01157","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["combra-lab/spiking-ddpg-mapless-navigation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","n_code_links":4,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improved-exploration-through-latent","title":"Improved Exploration through Latent Trajectory Optimization in Deep Deterministic Policy Gradient","date":"2019-11-15","arxiv_id":"1911.06833","n_code_links":0,"syntology":null},{"paper":null,"slug":"ctrl-z-recovering-from-instability-in","title":"Ctrl-Z: Recovering from Instability in Reinforcement Learning","date":"2019-10-09","arxiv_id":"1910.03732","n_code_links":0,"syntology":null},{"paper":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"constrained-attractor-selection-using-deep","title":"Constrained Attractor Selection Using Deep Reinforcement Learning","date":"2019-09-23","arxiv_id":"1909.10500","n_code_links":0,"syntology":null},{"paper":"/paper/ac-teach-a-bayesian-actor-critic-method-for","slug":"ac-teach-a-bayesian-actor-critic-method-for","title":"AC-Teach: A Bayesian Actor-Critic Method for Policy Learning with an Ensemble of Suboptimal Teachers","date":"2019-09-09","arxiv_id":"1909.04121","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":null}},{"paper":null,"slug":"deterministic-value-policy-gradients","title":"Deterministic Value-Policy Gradients","date":"2019-09-09","arxiv_id":"1909.03939","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-control-for-bipedal-locomotion","slug":"hierarchical-control-for-bipedal-locomotion","title":"Hierarchical Control for Bipedal Locomotion using Central Pattern Generators and Neural Networks","date":"2019-09-02","arxiv_id":"1909.00732","n_code_links":1,"syntology":null},{"paper":null,"slug":"incremental-reinforcement-learning-a-new","title":"Incremental Reinforcement Learning --- a New Continuous Reinforcement Learning Frame Based on Stochastic Differential Equation methods","date":"2019-08-08","arxiv_id":"1908.02974","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-reinforcement-learning-through","title":"Improved Reinforcement Learning through Imitation Learning Pretraining Towards Image-based Autonomous Driving","date":"2019-07-16","arxiv_id":"1907.06838","n_code_links":0,"syntology":null},{"paper":"/paper/rethink-global-reward-game-and-credit","slug":"rethink-global-reward-game-and-credit","title":"Shapley Q-value: A Local Reward Approach to Solve Global Reward Games","date":"2019-07-11","arxiv_id":"1907.05707","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["hsvgbkhgbv/SQDDPG"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-for-unmanned","title":"Deep Reinforcement Learning for Unmanned Aerial Vehicle-Assisted Vehicular Networks","date":"2019-06-12","arxiv_id":"1906.05015","n_code_links":0,"syntology":null},{"paper":"/paper/190600214","slug":"190600214","title":"Harnessing Reinforcement Learning for Neural Motion Planning","date":"2019-06-01","arxiv_id":"1906.00214","n_code_links":1,"syntology":null},{"paper":"/paper/deep-residual-reinforcement-learning","slug":"deep-residual-reinforcement-learning","title":"Deep Residual Reinforcement Learning","date":"2019-05-03","arxiv_id":"1905.01072","n_code_links":1,"syntology":null},{"paper":null,"slug":"cem-rl-combining-evolutionary-and-gradient","title":"CEM-RL: Combining evolutionary and gradient-based methods for policy search","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-agents-with-prioritization-and","title":"Learning agents with prioritization and parameter noise in continuous state and action space","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-feedback","slug":"deep-reinforcement-learning-with-feedback","title":"Deep Reinforcement Learning with Feedback-based Exploration","date":"2019-03-14","arxiv_id":"1903.06151","n_code_links":2,"syntology":null},{"paper":"/paper/asynchronous-episodic-deep-deterministic","slug":"asynchronous-episodic-deep-deterministic","title":"Asynchronous Episodic Deep Deterministic Policy Gradient: Towards Continuous Control in Computationally Complex Environments","date":"2019-03-03","arxiv_id":"1903.00827","n_code_links":1,"syntology":null},{"paper":"/paper/crossnorm-normalization-for-off-policy-td","slug":"crossnorm-normalization-for-off-policy-td","title":"CrossQ: Batch Normalization in Deep Reinforcement Learning for Greater Sample Efficiency and Simplicity","date":"2019-02-14","arxiv_id":"1902.05605","n_code_links":5,"syntology":null},{"paper":null,"slug":"reward-shaping-via-meta-learning","title":"Reward Shaping via Meta-Learning","date":"2019-01-27","arxiv_id":"1901.09330","n_code_links":0,"syntology":null},{"paper":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","n_code_links":2,"syntology":null},{"paper":"/paper/transfer-learning-for-prosthetics-using","slug":"transfer-learning-for-prosthetics-using","title":"Transfer Learning for Prosthetics Using Imitation Learning","date":"2019-01-15","arxiv_id":"1901.04772","n_code_links":1,"syntology":null},{"paper":"/paper/decentralized-computation-offloading-for","slug":"decentralized-computation-offloading-for","title":"Decentralized Computation Offloading for Multi-User Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2018-12-16","arxiv_id":"1812.07394","n_code_links":2,"syntology":null},{"paper":"/paper/off-policy-deep-reinforcement-learning","slug":"off-policy-deep-reinforcement-learning","title":"Off-Policy Deep Reinforcement Learning without Exploration","date":"2018-12-07","arxiv_id":"1812.02900","n_code_links":10,"syntology":{"ran":14,"of":14,"n_ran_checked":14,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"14 ran (of which 12 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sfujim/BCQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"resource-constrained-deep-reinforcement","title":"Resource Constrained Deep Reinforcement Learning","date":"2018-12-03","arxiv_id":"1812.00600","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-autonomous","slug":"deep-reinforcement-learning-for-autonomous","title":"Deep Reinforcement Learning for Autonomous Driving","date":"2018-11-28","arxiv_id":"1811.11329","n_code_links":1,"syntology":null},{"paper":null,"slug":"modelling-the-dynamic-joint-policy-of","title":"Modelling the Dynamic Joint Policy of Teammates with Attention Multi-agent DDPG","date":"2018-11-13","arxiv_id":"1811.07029","n_code_links":0,"syntology":null},{"paper":"/paper/ace-an-actor-ensemble-algorithm-for","slug":"ace-an-actor-ensemble-algorithm-for","title":"ACE: An Actor Ensemble Algorithm for Continuous Control with Tree Search","date":"2018-11-06","arxiv_id":"1811.02696","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-approaches-for-reinforcement","title":"Hierarchical Approaches for Reinforcement Learning in Parameterized Action Space","date":"2018-10-23","arxiv_id":"1810.09656","n_code_links":0,"syntology":null},{"paper":"/paper/curious-intrinsically-motivated-multi-task","slug":"curious-intrinsically-motivated-multi-task","title":"CURIOUS: Intrinsically Motivated Modular Multi-Goal Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06284","n_code_links":1,"syntology":null},{"paper":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","n_code_links":5,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"curriculum-goal-masking-for-continuous-deep","title":"Curriculum goal masking for continuous deep reinforcement learning","date":"2018-09-17","arxiv_id":"1809.06146","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-deep-reinforcement-learning-in","slug":"adversarial-deep-reinforcement-learning-in","title":"Adversarial Deep Reinforcement Learning in Portfolio Management","date":"2018-08-29","arxiv_id":"1808.09940","n_code_links":5,"syntology":null}],"record_sha256":"22755f058ae52ad56883d7c611fef696c1c0c467b2ca73a710ef2303209e3e10","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}