{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/8","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":8,"pages_in_order":9,"rows_per_page":100,"rows":[701,800],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay/papers/7","next":"/method/experience-replay/papers/9","papers":[{"paper":null,"slug":"learning-sparse-representations-incrementally","title":"Learning Sparse Representations Incrementally in Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.04002","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-policy-reinforcement-learning-with-entropy","title":"Policy Optimization Reinforcement Learning with Entropy Regularization","date":"2019-12-02","arxiv_id":"1912.01557","n_code_links":0,"syntology":null},{"paper":"/paper/better-exploration-with-optimistic-actor-1","slug":"better-exploration-with-optimistic-actor-1","title":"Better Exploration with Optimistic Actor Critic","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/curriculum-guided-hindsight-experience-replay","slug":"curriculum-guided-hindsight-experience-replay","title":"Curriculum-guided Hindsight Experience Replay","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/learning-reward-machines-for-partially","slug":"learning-reward-machines-for-partially","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/online-continual-learning-with-maximal","slug":"online-continual-learning-with-maximal","title":"Online Continual Learning with Maximal Interfered Retrieval","date":"2019-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/reconciling-returns-with-experience-replay","slug":"reconciling-returns-with-experience-replay","title":"Reconciling λ-Returns with Experience Replay","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"impact-importance-weighted-asynchronous-1","title":"IMPACT: Importance Weighted Asynchronous Architectures with Clipped Target Networks","date":"2019-11-30","arxiv_id":"1912.00167","n_code_links":0,"syntology":null},{"paper":"/paper/distributed-soft-actor-critic-with","slug":"distributed-soft-actor-critic-with","title":"Distributed Soft Actor-Critic with Multivariate Reward Representation and Knowledge Distillation","date":"2019-11-29","arxiv_id":"1911.13056","n_code_links":1,"syntology":null},{"paper":null,"slug":"which-channel-to-ask-my-question-personalized","title":"Which Channel to Ask My Question? Personalized Customer Service RequestStream Routing using DeepReinforcement Learning","date":"2019-11-24","arxiv_id":"1911.10521","n_code_links":0,"syntology":null},{"paper":null,"slug":"placement-optimization-of-aerial-base","title":"Placement Optimization of Aerial Base Stations with Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08111","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-efficient-co-adaptation-of-morphology","title":"Data-efficient Co-Adaptation of Morphology and Behaviour with Deep Reinforcement Learning","date":"2019-11-15","arxiv_id":"1911.06832","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-exploration-through-latent","title":"Improved Exploration through Latent Trajectory Optimization in Deep Deterministic Policy Gradient","date":"2019-11-15","arxiv_id":"1911.06833","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-dynamic","title":"Deep Reinforcement Learning Based Dynamic Trajectory Control for UAV-assisted Mobile Edge Computing","date":"2019-11-10","arxiv_id":"1911.03887","n_code_links":0,"syntology":null},{"paper":"/paper/policy-continuation-with-hindsight-inverse","slug":"policy-continuation-with-hindsight-inverse","title":"Policy Continuation with Hindsight Inverse Dynamics","date":"2019-10-30","arxiv_id":"1910.14055","n_code_links":1,"syntology":null},{"paper":null,"slug":"191013213","title":"Overcoming Catastrophic Interference in Online Reinforcement Learning with Dynamic Self-Organizing Maps","date":"2019-10-29","arxiv_id":"1910.13213","n_code_links":0,"syntology":null},{"paper":null,"slug":"better-exploration-with-optimistic-actor","title":"Better Exploration with Optimistic Actor-Critic","date":"2019-10-28","arxiv_id":"1910.12807","n_code_links":0,"syntology":null},{"paper":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-educated-language-agent-with-hindsight-1","title":"HIGhER : Improving instruction following with Hindsight Generation for Experience Replay","date":"2019-10-21","arxiv_id":"1910.09451","n_code_links":0,"syntology":null},{"paper":null,"slug":"reverse-experience-replay","title":"Reverse Experience Replay","date":"2019-10-19","arxiv_id":"1910.08780","n_code_links":0,"syntology":null},{"paper":"/paper/towards-more-sample-efficiency","slug":"towards-more-sample-efficiency","title":"Towards More Sample Efficiency in Reinforcement Learning with Data Augmentation","date":"2019-10-19","arxiv_id":"1910.09959","n_code_links":1,"syntology":null},{"paper":null,"slug":"ctrl-z-recovering-from-instability-in","title":"Ctrl-Z: Recovering from Instability in Reinforcement Learning","date":"2019-10-09","arxiv_id":"1910.03732","n_code_links":0,"syntology":null},{"paper":"/paper/torchbeast-a-pytorch-platform-for-distributed","slug":"torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","n_code_links":3,"syntology":{"ran":6,"of":9,"n_ran_checked":2,"n_instrument":4,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["heiner/scalable_agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/towards-simplicity-in-deep-reinforcement-1","slug":"towards-simplicity-in-deep-reinforcement-1","title":"Striving for Simplicity and Performance in Off-Policy DRL: Output Normalization and Non-Uniform Sampling","date":"2019-10-05","arxiv_id":"1910.02208","n_code_links":3,"syntology":null},{"paper":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"off-policy-multi-step-q-learning","title":"Composite Q-learning: Multi-scale Q-function Decomposition and Separable Optimization","date":"2019-09-30","arxiv_id":"1909.13518","n_code_links":0,"syntology":null},{"paper":"/paper/off-policy-actor-critic-with-shared-1","slug":"off-policy-actor-critic-with-shared-1","title":"Off-Policy Actor-Critic with Shared Experience Replay","date":"2019-09-25","arxiv_id":"1909.11583","n_code_links":0,"syntology":null},{"paper":"/paper/invariant-transform-experience-replay","slug":"invariant-transform-experience-replay","title":"Invariant Transform Experience Replay: Data Augmentation for Deep Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.10707","n_code_links":1,"syntology":null},{"paper":null,"slug":"constrained-attractor-selection-using-deep","title":"Constrained Attractor Selection Using Deep Reinforcement Learning","date":"2019-09-23","arxiv_id":"1909.10500","n_code_links":0,"syntology":null},{"paper":"/paper/ac-teach-a-bayesian-actor-critic-method-for","slug":"ac-teach-a-bayesian-actor-critic-method-for","title":"AC-Teach: A Bayesian Actor-Critic Method for Policy Learning with an Ensemble of Suboptimal Teachers","date":"2019-09-09","arxiv_id":"1909.04121","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":null}},{"paper":null,"slug":"deterministic-value-policy-gradients","title":"Deterministic Value-Policy Gradients","date":"2019-09-09","arxiv_id":"1909.03939","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-for-control-of","slug":"deep-reinforcement-learning-for-control-of","title":"Deep Reinforcement Learning for Control of Probabilistic Boolean Networks","date":"2019-09-07","arxiv_id":"1909.03331","n_code_links":2,"syntology":null},{"paper":null,"slug":"efficient-automatic-meta-optimization-search","title":"Efficient Automatic Meta Optimization Search for Few-Shot Learning","date":"2019-09-06","arxiv_id":"1909.03817","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-control-for-bipedal-locomotion","slug":"hierarchical-control-for-bipedal-locomotion","title":"Hierarchical Control for Bipedal Locomotion using Central Pattern Generators and Neural Networks","date":"2019-09-02","arxiv_id":"1909.00732","n_code_links":1,"syntology":null},{"paper":null,"slug":"iterative-update-and-unified-representation","title":"Iterative Update and Unified Representation for Multi-Agent Reinforcement Learning","date":"2019-08-16","arxiv_id":"1908.06758","n_code_links":0,"syntology":null},{"paper":"/paper/online-continual-learning-with-maximally","slug":"online-continual-learning-with-maximally","title":"Online Continual Learning with Maximally Interfered Retrieval","date":"2019-08-11","arxiv_id":"1908.04742","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":6,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["optimass/Maximally_Interfered_Retrieval"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"incremental-reinforcement-learning-a-new","title":"Incremental Reinforcement Learning --- a New Continuous Reinforcement Learning Frame Based on Stochastic Differential Equation methods","date":"2019-08-08","arxiv_id":"1908.02974","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-control-with-metric-learning","title":"Attention Control with Metric Learning Alignment for Image Set-based Recognition","date":"2019-08-05","arxiv_id":"1908.01872","n_code_links":0,"syntology":null},{"paper":null,"slug":"prioritized-guidance-for-efficient-multi","title":"Prioritized Guidance for Efficient Multi-Agent Reinforcement Learning Exploration","date":"2019-07-18","arxiv_id":"1907.07847","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-reinforcement-learning-through","title":"Improved Reinforcement Learning through Imitation Learning Pretraining Towards Image-based Autonomous Driving","date":"2019-07-16","arxiv_id":"1907.06838","n_code_links":0,"syntology":null},{"paper":"/paper/rethink-global-reward-game-and-credit","slug":"rethink-global-reward-game-and-credit","title":"Shapley Q-value: A Local Reward Approach to Solve Global Reward Games","date":"2019-07-11","arxiv_id":"1907.05707","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["hsvgbkhgbv/SQDDPG"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dependency-aware-attention-control-for-1","title":"Dependency-aware Attention Control for Unconstrained Face Recognition with Image Sets","date":"2019-07-05","arxiv_id":"1907.03030","n_code_links":0,"syntology":null},{"paper":null,"slug":"modified-actor-critics","title":"Modified Actor-Critics","date":"2019-07-02","arxiv_id":"1907.01298","n_code_links":0,"syntology":null},{"paper":"/paper/variational-quantum-circuits-and-deep","slug":"variational-quantum-circuits-and-deep","title":"Variational Quantum Circuits for Deep Reinforcement Learning","date":"2019-06-30","arxiv_id":"1907.00397","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimal-use-of-experience-in-first-person","title":"Optimal Use of Experience in First Person Shooter Environments","date":"2019-06-24","arxiv_id":"1906.09734","n_code_links":0,"syntology":null},{"paper":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-model-based-planning-with-policy","slug":"exploring-model-based-planning-with-policy","title":"Exploring Model-based Planning with Policy Networks","date":"2019-06-20","arxiv_id":"1906.08649","n_code_links":1,"syntology":null},{"paper":null,"slug":"experience-replay-optimization","title":"Experience Replay Optimization","date":"2019-06-19","arxiv_id":"1906.08387","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolutionary-reinforcement-learning-for","title":"Evolutionary Reinforcement Learning for Sample-Efficient Multiagent Coordination","date":"2019-06-18","arxiv_id":"1906.07315","n_code_links":0,"syntology":null},{"paper":"/paper/goal-conditioned-imitation-learning","slug":"goal-conditioned-imitation-learning","title":"Goal-conditioned Imitation Learning","date":"2019-06-13","arxiv_id":"1906.05838","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dingyiming0427/goalgail"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-for-unmanned","title":"Deep Reinforcement Learning for Unmanned Aerial Vehicle-Assisted Vehicular Networks","date":"2019-06-12","arxiv_id":"1906.05015","n_code_links":0,"syntology":null},{"paper":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/exploration-via-hindsight-goal-generation","slug":"exploration-via-hindsight-goal-generation","title":"Exploration via Hindsight Goal Generation","date":"2019-06-10","arxiv_id":"1906.04279","n_code_links":1,"syntology":null},{"paper":"/paper/improving-exploration-in-soft-actor-critic","slug":"improving-exploration-in-soft-actor-critic","title":"Improving Exploration in Soft-Actor-Critic with Normalizing Flows Policies","date":"2019-06-06","arxiv_id":"1906.02771","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["joeybose/FloRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploration-with-unreliable-intrinsic-reward","title":"Exploration with Unreliable Intrinsic Reward in Multi-Agent Reinforcement Learning","date":"2019-06-05","arxiv_id":"1906.02138","n_code_links":0,"syntology":null},{"paper":"/paper/episodic-memory-in-lifelong-language-learning","slug":"episodic-memory-in-lifelong-language-learning","title":"Episodic Memory in Lifelong Language Learning","date":"2019-06-03","arxiv_id":"1906.01076","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/190600214","slug":"190600214","title":"Harnessing Reinforcement Learning for Neural Motion Planning","date":"2019-06-01","arxiv_id":"1906.00214","n_code_links":1,"syntology":null},{"paper":null,"slug":"190512726","title":"Prioritized Sequence Experience Replay","date":"2019-05-25","arxiv_id":"1905.12726","n_code_links":0,"syntology":null},{"paper":"/paper/maximum-entropy-regularized-multi-goal","slug":"maximum-entropy-regularized-multi-goal","title":"Maximum Entropy-Regularized Multi-Goal Reinforcement Learning","date":"2019-05-21","arxiv_id":"1905.08786","n_code_links":3,"syntology":null},{"paper":"/paper/combining-experience-replay-with-exploration","slug":"combining-experience-replay-with-exploration","title":"Combining Experience Replay with Exploration by Random Network Distillation","date":"2019-05-18","arxiv_id":"1905.07579","n_code_links":1,"syntology":null},{"paper":"/paper/bias-reduced-hindsight-experience-replay-with","slug":"bias-reduced-hindsight-experience-replay-with","title":"Bias-Reduced Hindsight Experience Replay with Virtual Goal Prioritization","date":"2019-05-14","arxiv_id":"1905.05498","n_code_links":1,"syntology":null},{"paper":"/paper/deep-residual-reinforcement-learning","slug":"deep-residual-reinforcement-learning","title":"Deep Residual Reinforcement Learning","date":"2019-05-03","arxiv_id":"1905.01072","n_code_links":1,"syntology":null},{"paper":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"actrce-augmenting-experience-via-teachers-1","title":"ACTRCE: Augmenting Experience via Teacher’s Advice","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cem-rl-combining-evolutionary-and-gradient","title":"CEM-RL: Combining evolutionary and gradient-based methods for policy search","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/dher-hindsight-experience-replay-for-dynamic","slug":"dher-hindsight-experience-replay-for-dynamic","title":"DHER: Hindsight Experience Replay for Dynamic Goals","date":"2019-05-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-agents-with-prioritization-and","title":"Learning agents with prioritization and parameter noise in continuous state and action space","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-goal-conditioned-value-functions","title":"Learning Goal-Conditioned Value Functions with one-step Path rewards rather than Goal-Rewards","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-combining-on-off-policy-methods-for","title":"Towards Combining On-Off-Policy Methods for Real-World Applications","date":"2019-04-24","arxiv_id":"1904.10642","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-feedback","slug":"deep-reinforcement-learning-with-feedback","title":"Deep Reinforcement Learning with Feedback-based Exploration","date":"2019-03-14","arxiv_id":"1903.06151","n_code_links":2,"syntology":null},{"paper":null,"slug":"complementary-learning-for-overcoming","title":"Complementary Learning for Overcoming Catastrophic Forgetting Using Experience Replay","date":"2019-03-11","arxiv_id":"1903.04566","n_code_links":0,"syntology":null},{"paper":"/paper/asynchronous-episodic-deep-deterministic","slug":"asynchronous-episodic-deep-deterministic","title":"Asynchronous Episodic Deep Deterministic Policy Gradient: Towards Continuous Control in Computationally Complex Environments","date":"2019-03-03","arxiv_id":"1903.00827","n_code_links":1,"syntology":null},{"paper":"/paper/190504100","slug":"190504100","title":"Deep Reinforcement Learning using Genetic Algorithm for Parameter Optimization","date":"2019-02-19","arxiv_id":"1905.04100","n_code_links":2,"syntology":{"ran":9,"of":13,"n_ran_checked":7,"n_instrument":2,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["aralab-unr/ReinforcementLearningWithGA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/crossnorm-normalization-for-off-policy-td","slug":"crossnorm-normalization-for-off-policy-td","title":"CrossQ: Batch Normalization in Deep Reinforcement Learning for Greater Sample Efficiency and Simplicity","date":"2019-02-14","arxiv_id":"1902.05605","n_code_links":5,"syntology":null},{"paper":null,"slug":"actrce-augmenting-experience-via-teachers","title":"ACTRCE: Augmenting Experience via Teacher's Advice For Multi-Goal Reinforcement Learning","date":"2019-02-12","arxiv_id":"1902.04546","n_code_links":0,"syntology":null},{"paper":null,"slug":"competitive-experience-replay","title":"Competitive Experience Replay","date":"2019-02-01","arxiv_id":"1902.00528","n_code_links":0,"syntology":null},{"paper":"/paper/visual-hindsight-experience-replay","slug":"visual-hindsight-experience-replay","title":"Addressing Sample Complexity in Visual Tasks Using HER and Hallucinatory GANs","date":"2019-01-31","arxiv_id":"1901.11529","n_code_links":2,"syntology":null},{"paper":null,"slug":"reward-shaping-via-meta-learning","title":"Reward Shaping via Meta-Learning","date":"2019-01-27","arxiv_id":"1901.09330","n_code_links":0,"syntology":null},{"paper":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","n_code_links":2,"syntology":null},{"paper":"/paper/transfer-learning-for-prosthetics-using","slug":"transfer-learning-for-prosthetics-using","title":"Transfer Learning for Prosthetics Using Imitation Learning","date":"2019-01-15","arxiv_id":"1901.04772","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-theoretical-analysis-of-deep-q-learning","title":"A Theoretical Analysis of Deep Q-Learning","date":"2019-01-01","arxiv_id":"1901.00137","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-deep-q-learning-for-optimal-execution","title":"Double Deep Q-Learning for Optimal Execution","date":"2018-12-17","arxiv_id":"1812.06600","n_code_links":0,"syntology":null},{"paper":"/paper/decentralized-computation-offloading-for","slug":"decentralized-computation-offloading-for","title":"Decentralized Computation Offloading for Multi-User Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2018-12-16","arxiv_id":"1812.07394","n_code_links":2,"syntology":null},{"paper":"/paper/soft-actor-critic-algorithms-and-applications","slug":"soft-actor-critic-algorithms-and-applications","title":"Soft Actor-Critic Algorithms and Applications","date":"2018-12-13","arxiv_id":"1812.05905","n_code_links":52,"syntology":{"ran":38,"of":40,"n_ran_checked":35,"n_instrument":3,"unverified":2,"pointer_only":5,"phrase":"38 ran (of which 0 constructed an object rather than computing a result; 35 with no instrument failure: 4 honoured, 1 violated, 30 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["rail-berkeley/softlearning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/off-policy-deep-reinforcement-learning","slug":"off-policy-deep-reinforcement-learning","title":"Off-Policy Deep Reinforcement Learning without Exploration","date":"2018-12-07","arxiv_id":"1812.02900","n_code_links":10,"syntology":{"ran":14,"of":14,"n_ran_checked":14,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"14 ran (of which 12 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sfujim/BCQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"deep-reinforcement-learning-and-the-deadly","title":"Deep Reinforcement Learning and the Deadly Triad","date":"2018-12-06","arxiv_id":"1812.02648","n_code_links":0,"syntology":null},{"paper":null,"slug":"resource-constrained-deep-reinforcement","title":"Resource Constrained Deep Reinforcement Learning","date":"2018-12-03","arxiv_id":"1812.00600","n_code_links":0,"syntology":null},{"paper":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","n_code_links":1,"syntology":{"ran":1,"of":7,"n_ran_checked":1,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/deep-reinforcement-learning-for-autonomous","slug":"deep-reinforcement-learning-for-autonomous","title":"Deep Reinforcement Learning for Autonomous Driving","date":"2018-11-28","arxiv_id":"1811.11329","n_code_links":1,"syntology":null},{"paper":null,"slug":"experience-replay-for-continual-learning","title":"Experience Replay for Continual Learning","date":"2018-11-28","arxiv_id":"1811.11682","n_code_links":0,"syntology":null},{"paper":null,"slug":"modelling-the-dynamic-joint-policy-of","title":"Modelling the Dynamic Joint Policy of Teammates with Attention Multi-agent DDPG","date":"2018-11-13","arxiv_id":"1811.07029","n_code_links":0,"syntology":null},{"paper":"/paper/ace-an-actor-ensemble-algorithm-for","slug":"ace-an-actor-ensemble-algorithm-for","title":"ACE: An Actor Ensemble Algorithm for Continuous Control with Tree Search","date":"2018-11-06","arxiv_id":"1811.02696","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-learn-without-forgetting-by","slug":"learning-to-learn-without-forgetting-by","title":"Learning to Learn without Forgetting by Maximizing Transfer and Minimizing Interference","date":"2018-10-29","arxiv_id":"1810.11910","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mattriemer/mer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/efficient-eligibility-traces-for-deep","slug":"efficient-eligibility-traces-for-deep","title":"Reconciling $λ$-Returns with Experience Replay","date":"2018-10-23","arxiv_id":"1810.09967","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-approaches-for-reinforcement","title":"Hierarchical Approaches for Reinforcement Learning in Parameterized Action Space","date":"2018-10-23","arxiv_id":"1810.09656","n_code_links":0,"syntology":null},{"paper":"/paper/curious-intrinsically-motivated-multi-task","slug":"curious-intrinsically-motivated-multi-task","title":"CURIOUS: Intrinsically Motivated Modular Multi-Goal Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06284","n_code_links":1,"syntology":null},{"paper":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","n_code_links":5,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/energy-based-hindsight-experience","slug":"energy-based-hindsight-experience","title":"Energy-Based Hindsight Experience Prioritization","date":"2018-10-02","arxiv_id":"1810.01363","n_code_links":2,"syntology":null},{"paper":null,"slug":"hierarchical-deep-multiagent-reinforcement","title":"Hierarchical Deep Multiagent Reinforcement Learning with Temporal Abstraction","date":"2018-09-25","arxiv_id":"1809.09332","n_code_links":0,"syntology":null}],"record_sha256":"fa1856615ca98b9ae94ebe8099f152540bc447c378d5a67f832b46772ac2a4a0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}