{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/6","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":9,"rows_per_page":100,"rows":[501,600],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay/papers/5","next":"/method/experience-replay/papers/7","papers":[{"paper":null,"slug":"learning-expected-emphatic-traces-for-deep-rl","title":"Learning Expected Emphatic Traces for Deep RL","date":"2021-07-12","arxiv_id":"2107.05405","n_code_links":0,"syntology":null},{"paper":null,"slug":"mher-model-based-hindsight-experience-replay","title":"MHER: Model-based Hindsight Experience Replay","date":"2021-07-01","arxiv_id":"2107.00306","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-based-secure-noma-and-cognitive-radio","title":"AI-Based Secure NOMA and Cognitive Radio enabled Green Communications: Channel State Information and Battery Value Uncertainties","date":"2021-06-30","arxiv_id":"2106.15964","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-continual-learning-via-self","title":"Unsupervised Continual Learning via Self-Adaptive Deep Clustering Approach","date":"2021-06-28","arxiv_id":"2106.14563","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-reinforcement-learning-approach-for-an-irs","title":"A Reinforcement Learning Approach for an IRS-assisted NOMA Network","date":"2021-06-17","arxiv_id":"2106.09611","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-1","title":"Deep Reinforcement Learning Based Optimization for IRS Based UAV-NOMA Downlink Networks","date":"2021-06-17","arxiv_id":"2106.09616","n_code_links":0,"syntology":null},{"paper":null,"slug":"many-agent-reinforcement-learning-under","title":"Many Agent Reinforcement Learning Under Partial Observability","date":"2021-06-17","arxiv_id":"2106.09825","n_code_links":0,"syntology":null},{"paper":null,"slug":"unbiased-methods-for-multi-goal-reinforcement","title":"Unbiased Methods for Multi-Goal Reinforcement Learning","date":"2021-06-16","arxiv_id":"2106.08863","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-continuous-control-with-double","slug":"efficient-continuous-control-with-double","title":"Efficient Continuous Control with Double Actors and Regularized Critics","date":"2021-06-06","arxiv_id":"2106.03050","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-uav","title":"Deep Reinforcement Learning-based UAV Navigation and Control: A Soft Actor-Critic with Hindsight Experience Replay Approach","date":"2021-06-02","arxiv_id":"2106.01016","n_code_links":0,"syntology":null},{"paper":null,"slug":"variational-empowerment-as-representation","title":"Variational Empowerment as Representation Learning for Goal-Based Reinforcement Learning","date":"2021-06-02","arxiv_id":"2106.01404","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-intrinsic-motivation-for","slug":"adversarial-intrinsic-motivation-for","title":"Adversarial Intrinsic Motivation for Reinforcement Learning","date":"2021-05-27","arxiv_id":"2105.13345","n_code_links":1,"syntology":{"ran":10,"of":14,"n_ran_checked":10,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["iDurugkar/adversarial-intrinsic-motivation"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-optimize-industry-scale-dynamic","title":"Learning to Optimize Industry-Scale Dynamic Pickup and Delivery Problems","date":"2021-05-27","arxiv_id":"2105.12899","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimistic-reinforcement-learning-by-forward","title":"Optimistic Reinforcement Learning by Forward Kullback-Leibler Divergence Optimization","date":"2021-05-27","arxiv_id":"2105.12991","n_code_links":0,"syntology":null},{"paper":null,"slug":"near-optimal-offline-and-streaming-algorithms","title":"Near-optimal Offline and Streaming Algorithms for Learning Non-Linear Dynamical Systems","date":"2021-05-24","arxiv_id":"2105.11558","n_code_links":0,"syntology":null},{"paper":"/paper/improved-exploring-starts-by-kernel-density","slug":"improved-exploring-starts-by-kernel-density","title":"Improved Exploring Starts by Kernel Density Estimation-Based State-Space Coverage Acceleration in Reinforcement Learning","date":"2021-05-19","arxiv_id":"2105.08990","n_code_links":1,"syntology":null},{"paper":null,"slug":"make-bipedal-robots-learn-how-to-imitate","title":"Make Bipedal Robots Learn How to Imitate","date":"2021-05-15","arxiv_id":"2105.07193","n_code_links":0,"syntology":null},{"paper":"/paper/regret-minimization-experience-replay","slug":"regret-minimization-experience-replay","title":"Regret Minimization Experience Replay in Off-Policy Reinforcement Learning","date":"2021-05-15","arxiv_id":"2105.07253","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-rnns-based-transformers-maddpg","title":"Hierarchical RNNs-Based Transformers MADDPG for Mixed Cooperative-Competitive Environments","date":"2021-05-11","arxiv_id":"2105.04888","n_code_links":0,"syntology":null},{"paper":"/paper/context-based-soft-actor-critic-for","slug":"context-based-soft-actor-critic-for","title":"Context-Based Soft Actor Critic for Environments with Non-stationary Dynamics","date":"2021-05-07","arxiv_id":"2105.03310","n_code_links":1,"syntology":null},{"paper":null,"slug":"time-aware-q-networks-resolving-temporal","title":"Time-Aware Q-Networks: Resolving Temporal Irregularity for Deep Reinforcement Learning","date":"2021-05-06","arxiv_id":"2105.02580","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-grasping-policies-for-human-in-the","slug":"end-to-end-grasping-policies-for-human-in-the","title":"End-to-end grasping policies for human-in-the-loop robots via deep reinforcement learning","date":"2021-04-26","arxiv_id":"2104.12842","n_code_links":1,"syntology":null},{"paper":"/paper/class-incremental-experience-replay-for","slug":"class-incremental-experience-replay-for","title":"Class-Incremental Experience Replay for Continual Learning under Concept Drift","date":"2021-04-24","arxiv_id":"2104.11861","n_code_links":1,"syntology":null},{"paper":"/paper/independent-reinforcement-learning-for-weakly","slug":"independent-reinforcement-learning-for-weakly","title":"Independent Reinforcement Learning for Weakly Cooperative Multiagent Traffic Control Problem","date":"2021-04-22","arxiv_id":"2104.10917","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-deterministic-path-following","title":"Deep Deterministic Path Following","date":"2021-04-13","arxiv_id":"2104.06014","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-controller","title":"Deep Reinforcement Learning Based Controller for Active Heave Compensation","date":"2021-04-12","arxiv_id":"2104.05599","n_code_links":0,"syntology":null},{"paper":"/paper/reducing-representation-drift-in-online","slug":"reducing-representation-drift-in-online","title":"New Insights on Reducing Abrupt Representation Change in Online Continual Learning","date":"2021-04-11","arxiv_id":"2104.05025","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["naderAsadi/AML","pclucas14/aml"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"progressive-extension-of-reinforcement","title":"Progressive extension of reinforcement learning action dimension for asymmetric assembly tasks","date":"2021-04-06","arxiv_id":"2104.04078","n_code_links":0,"syntology":null},{"paper":"/paper/self-adaptive-torque-vectoring-controller","slug":"self-adaptive-torque-vectoring-controller","title":"Self-adaptive Torque Vectoring Controller Using Reinforcement Learning","date":"2021-03-27","arxiv_id":"2103.14892","n_code_links":1,"syntology":null},{"paper":null,"slug":"continual-speaker-adaptation-for-text-to","title":"Continual Speaker Adaptation for Text-to-Speech Synthesis","date":"2021-03-26","arxiv_id":"2103.14512","n_code_links":0,"syntology":null},{"paper":"/paper/tf-gczsl-task-free-generalized-continual-zero","slug":"tf-gczsl-task-free-generalized-continual-zero","title":"Online Lifelong Generalized Zero-Shot Learning","date":"2021-03-19","arxiv_id":"2103.10741","n_code_links":1,"syntology":null},{"paper":null,"slug":"simulation-studies-on-deep-reinforcement","title":"Simulation Studies on Deep Reinforcement Learning for Building Control with Human Interaction","date":"2021-03-14","arxiv_id":"2103.07919","n_code_links":0,"syntology":null},{"paper":null,"slug":"streaming-linear-system-identification-with","title":"Streaming Linear System Identification with Reverse Experience Replay","date":"2021-03-10","arxiv_id":"2103.05896","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-play-soccer-from-scratch-sample","slug":"learning-to-play-soccer-from-scratch-sample","title":"Learning to Play Soccer From Scratch: Sample-Efficient Emergent Coordination through Curriculum-Learning and Competition","date":"2021-03-09","arxiv_id":"2103.05174","n_code_links":1,"syntology":null},{"paper":"/paper/selective-replay-enhances-learning-in-online","slug":"selective-replay-enhances-learning-in-online","title":"Selective Replay Enhances Learning in Online Continual Analogical Reasoning","date":"2021-03-06","arxiv_id":"2103.03987","n_code_links":1,"syntology":null},{"paper":"/paper/improving-computational-efficiency-in-visual","slug":"improving-computational-efficiency-in-visual","title":"Improving Computational Efficiency in Visual Reinforcement Learning via Stored Embeddings","date":"2021-03-04","arxiv_id":"2103.02886","n_code_links":1,"syntology":null},{"paper":null,"slug":"data-driven-mimo-control-of-room-temperature","title":"Data-driven control of room temperature and bidirectional EV charging using deep reinforcement learning: simulations and experiments","date":"2021-03-02","arxiv_id":"2103.01886","n_code_links":0,"syntology":null},{"paper":"/paper/a-deeppixbis-attentional-angular-margin-for","slug":"a-deeppixbis-attentional-angular-margin-for","title":"A-DeepPixBis: Attentional Angular Margin for Face Anti-Spoofing","date":"2021-03-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-path-planning-based-on-mpc-and","title":"Multi-Agent Path Planning based on MPC and DDPG","date":"2021-02-26","arxiv_id":"2102.13283","n_code_links":0,"syntology":null},{"paper":null,"slug":"application-of-twin-delayed-deep","title":"Twin actor twin delayed deep deterministic policy gradient (TATD3) learning for batch process control","date":"2021-02-25","arxiv_id":"2102.13012","n_code_links":0,"syntology":null},{"paper":null,"slug":"bias-reduced-multi-step-hindsight-experience","title":"Bias-reduced Multi-step Hindsight Experience Replay for Efficient Multi-goal Reinforcement Learning","date":"2021-02-25","arxiv_id":"2102.12962","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-regret-bound-and-experience-replay","title":"Improved Regret Bound and Experience Replay in Regularized Policy Iteration","date":"2021-02-25","arxiv_id":"2102.12611","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-car-following-strategy-based-on-deep","title":"Hybrid Car-Following Strategy based on Deep Deterministic Policy Gradient and Cooperative Adaptive Cruise Control","date":"2021-02-24","arxiv_id":"2103.03796","n_code_links":0,"syntology":null},{"paper":"/paper/memory-based-deep-reinforcement-learning-for-1","slug":"memory-based-deep-reinforcement-learning-for-1","title":"Memory-based Deep Reinforcement Learning for POMDPs","date":"2021-02-24","arxiv_id":"2102.12344","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["LinghengMeng/LSTM-TD3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"greedy-multi-step-off-policy-reinforcement-1","title":"Greedy-Step Off-Policy Reinforcement Learning","date":"2021-02-23","arxiv_id":"2102.11717","n_code_links":0,"syntology":null},{"paper":null,"slug":"escaping-from-zero-gradient-revisiting-action","title":"Escaping from Zero Gradient: Revisiting Action-Constrained Reinforcement Learning via Frank-Wolfe Policy Optimization","date":"2021-02-22","arxiv_id":"2102.11055","n_code_links":0,"syntology":null},{"paper":null,"slug":"stratified-experience-replay-correcting","title":"Stratified Experience Replay: Correcting Multiplicity Bias in Off-Policy Reinforcement Learning","date":"2021-02-22","arxiv_id":"2102.11319","n_code_links":0,"syntology":null},{"paper":"/paper/accelerated-sim-to-real-deep-reinforcement","slug":"accelerated-sim-to-real-deep-reinforcement","title":"Accelerated Sim-to-Real Deep Reinforcement Learning: Learning Collision Avoidance from Human Player","date":"2021-02-21","arxiv_id":"2102.10711","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-memory-dependent-continuous-control","title":"Learning Memory-Dependent Continuous Control from Demonstrations","date":"2021-02-18","arxiv_id":"2102.09208","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-with-symmetric","title":"Deep Reinforcement Learning with Symmetric Prior for Predictive Power Allocation to Mobile Users","date":"2021-02-10","arxiv_id":"2103.13298","n_code_links":0,"syntology":null},{"paper":"/paper/reverb-a-framework-for-experience-replay","slug":"reverb-a-framework-for-experience-replay","title":"Reverb: A Framework For Experience Replay","date":"2021-02-09","arxiv_id":"2102.04736","n_code_links":1,"syntology":null},{"paper":"/paper/explainable-reinforcement-learning-for","slug":"explainable-reinforcement-learning-for","title":"Explainable Reinforcement Learning for Longitudinal Control","date":"2021-02-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-prioritized-experience-replay-a-1","slug":"revisiting-prioritized-experience-replay-a-1","title":"Revisiting Prioritized Experience Replay: A Value Perspective","date":"2021-02-05","arxiv_id":"2102.03261","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-survey-of-motion-planning-algorithms-for","title":"A review of motion planning algorithms for intelligent robotics","date":"2021-02-04","arxiv_id":"2102.02376","n_code_links":0,"syntology":null},{"paper":null,"slug":"stay-alive-with-many-options-a-reinforcement","title":"Learning Skills to Navigate without a Master: A Sequential Multi-Policy Reinforcement Learning Algorithm","date":"2021-01-30","arxiv_id":"2102.00168","n_code_links":0,"syntology":null},{"paper":"/paper/offcon-3-what-is-state-of-the-art-anyway","slug":"offcon-3-what-is-state-of-the-art-anyway","title":"OffCon$^3$: What is state of the art anyway?","date":"2021-01-27","arxiv_id":"2101.11331","n_code_links":1,"syntology":null},{"paper":null,"slug":"gst-group-sparse-training-for-accelerating","title":"GST: Group-Sparse Training for Accelerating Deep Reinforcement Learning","date":"2021-01-24","arxiv_id":"2101.09650","n_code_links":0,"syntology":null},{"paper":"/paper/learning-synthetic-environments-for","slug":"learning-synthetic-environments-for","title":"Learning Synthetic Environments for Reinforcement Learning with Evolution Strategies","date":"2021-01-24","arxiv_id":"2101.09721","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-quantum","title":"Deep Reinforcement Learning with Quantum-inspired Experience Replay","date":"2021-01-06","arxiv_id":"2101.02034","n_code_links":0,"syntology":null},{"paper":null,"slug":"compute-and-memory-efficient-reinforcement","title":"Compute- and Memory-Efficient Reinforcement Learning with Latent Experience Replay","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"factored-action-spaces-in-deep-reinforcement","title":"Factored Action Spaces in Deep Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hellinger-distance-constrained-regression","title":"Hellinger Distance Constrained Regression","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hindsight-curriculum-generation-based-multi","title":"Hindsight Curriculum Generation Based Multi-Goal Experience Replay","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"online-continual-learning-under-domain-shift","title":"Online Continual Learning Under Domain Shift","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pgps-coupling-policy-gradient-with-population","title":"PGPS : Coupling Policy Gradient with Population-based Search","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"playing-atari-with-capsule-networks-a","title":"Playing Atari with Capsule Networks: A systematic comparison of CNN and CapsNets-based agents.","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-for-control-of-valves","slug":"reinforcement-learning-for-control-of-valves","title":"Reinforcement Learning for Control of Valves","date":"2020-12-29","arxiv_id":"2012.14668","n_code_links":2,"syntology":null},{"paper":null,"slug":"2012-11643","title":"myGym: Modular Toolkit for Visuomotor Robotic Tasks","date":"2020-12-21","arxiv_id":"2012.11643","n_code_links":0,"syntology":null},{"paper":null,"slug":"mobile-robots-autonomous-exploration-with","title":"Mobile Robots Autonomous Exploration with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-gradient-for-items-recommendation-on","title":"Policy Gradient for items Recommendation on Virtual Taobao","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","n_code_links":1,"syntology":null},{"paper":null,"slug":"ranking-items-in-large-scale-item-search","title":"Ranking Items in Large-Scale Item Search Engines with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"virtual-autonomous-driving-with-reinforcement","title":"Virtual Autonomous Driving with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"opac-opportunistic-actor-critic","title":"OPAC: Opportunistic Actor-Critic","date":"2020-12-11","arxiv_id":"2012.06555","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-reservoir-management-through-deep","title":"Efficient Reservoir Management through Deep Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.03822","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-correcting-q-learning","title":"Self-correcting Q-Learning","date":"2020-12-02","arxiv_id":"2012.01100","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-crowdsourced","title":"Deep Reinforcement Learning for Crowdsourced Urban Delivery: System States Characterization, Heuristics-guided Action Choice, and Rule-Interposing Integration","date":"2020-11-29","arxiv_id":"2011.14430","n_code_links":0,"syntology":null},{"paper":"/paper/episodic-self-imitation-learning-with","slug":"episodic-self-imitation-learning-with","title":"Episodic Self-Imitation Learning with Hindsight","date":"2020-11-26","arxiv_id":"2011.13467","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-from-simulation-racing-in-reality","title":"Learning from Simulation, Racing in Reality","date":"2020-11-26","arxiv_id":"2011.13332","n_code_links":0,"syntology":null},{"paper":null,"slug":"predictive-per-balancing-priority-and","title":"Predictive PER: Balancing Priority and Diversity towards Stable Deep Reinforcement Learning","date":"2020-11-26","arxiv_id":"2011.13093","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-robust-missile","title":"Reinforcement Learning for Robust Missile Autopilot Design","date":"2020-11-26","arxiv_id":"2011.12956","n_code_links":0,"syntology":null},{"paper":null,"slug":"consolidation-via-policy-information","title":"Consolidation via Policy Information Regularization in Deep RL for Multi-Agent Games","date":"2020-11-23","arxiv_id":"2011.11517","n_code_links":0,"syntology":null},{"paper":"/paper/finrl-a-deep-reinforcement-learning-library","slug":"finrl-a-deep-reinforcement-learning-library","title":"FinRL: A Deep Reinforcement Learning Library for Automated Stock Trading in Quantitative Finance","date":"2020-11-19","arxiv_id":"2011.09607","n_code_links":6,"syntology":null},{"paper":null,"slug":"acder-augmented-curiosity-driven-experience","title":"ACDER: Augmented Curiosity-Driven Experience Replay","date":"2020-11-16","arxiv_id":"2011.08027","n_code_links":0,"syntology":null},{"paper":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-deep-q-learning-based-path-planning-and","title":"A deep Q-Learning based Path Planning and Navigation System for Firefighting Environments","date":"2020-11-12","arxiv_id":"2011.06450","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-experiments-and","slug":"reinforcement-learning-experiments-and","title":"Reinforcement Learning Experiments and Benchmark for Solving Robotic Reaching Tasks","date":"2020-11-11","arxiv_id":"2011.05782","n_code_links":1,"syntology":null},{"paper":"/paper/realant-an-open-source-low-cost-quadruped-for","slug":"realant-an-open-source-low-cost-quadruped-for","title":"RealAnt: An Open-Source Low-Cost Quadruped for Education and Research in Real-World Reinforcement Learning","date":"2020-11-05","arxiv_id":"2011.03085","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-in-electricity","title":"Deep Reinforcement Learning in Electricity Generation Investment for the Minimization of Long-Term Carbon Emissions and Electricity Costs","date":"2020-11-02","arxiv_id":"2011.02342","n_code_links":0,"syntology":null},{"paper":"/paper/self-driving-network-and-service-coordination","slug":"self-driving-network-and-service-coordination","title":"Self-Driving Network and Service Coordination Using Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"optimizing-coverage-and-capacity-in-cellular","title":"Optimizing Coverage and Capacity in Cellular Networks using Machine Learning","date":"2020-10-22","arxiv_id":"2010.13710","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-surrogate-q-learning-for-autonomous","title":"Deep Surrogate Q-Learning for Autonomous Driving","date":"2020-10-21","arxiv_id":"2010.11278","n_code_links":0,"syntology":null},{"paper":null,"slug":"recurrent-distributed-reinforcement-learning","title":"A Learning Approach to Robot-Agnostic Force-Guided High Precision Assembly","date":"2020-10-15","arxiv_id":"2010.08052","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-experience-replay-a-bag-of-tricks","slug":"rethinking-experience-replay-a-bag-of-tricks","title":"Rethinking Experience Replay: a Bag of Tricks for Continual Learning","date":"2020-10-12","arxiv_id":"2010.05595","n_code_links":3,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hastings24/rethinking_er"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"deep-rl-with-information-constrained-policies","title":"Deep RL With Information Constrained Policies: Generalization in Continuous Control","date":"2020-10-09","arxiv_id":"2010.04646","n_code_links":0,"syntology":null},{"paper":null,"slug":"hindsight-experience-replay-with-kronecker","title":"Hindsight Experience Replay with Kronecker Product Approximate Curvature","date":"2020-10-09","arxiv_id":"2010.06142","n_code_links":0,"syntology":null},{"paper":"/paper/neural-mask-generator-learning-to-generate","slug":"neural-mask-generator-learning-to-generate","title":"Neural Mask Generator: Learning to Generate Adaptive Word Maskings for Language Model Adaptation","date":"2020-10-06","arxiv_id":"2010.02705","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-effectiveness-of-memory-replay-in-large-1","title":"The Effectiveness of Memory Replay in Large Scale Continual Learning","date":"2020-10-06","arxiv_id":"2010.02418","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-investigation-towards-efficient","slug":"an-empirical-investigation-towards-efficient","title":"An Empirical Investigation Towards Efficient Multi-Domain Language Model Pre-training","date":"2020-10-01","arxiv_id":"2010.00784","n_code_links":1,"syntology":null}],"record_sha256":"0379617e0b4cece48e13c8a5ab55bae656054bbed59b4943e5db31d67a1f0de4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}