{"url":"/method/r2d2","slug":"r2d2","name":"R2D2","full_name":"Recurrent Replay Distributed DQN","full_name_withheld":false,"description_markdown":"Building on the recent successes of distributed training of RL agents, R2D2 is an RL approach that trains a RNN-based RL agents from distributed prioritized experience replay. \r\nUsing a single network architecture and fixed set of hyperparameters, Recurrent Replay Distributed DQN quadrupled the previous state of the art on Atari-57, and matches the state of the art on DMLab-30. \r\nIt was the first agent to exceed human-level performance in 52 of the 57 Atari games.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Recurrent Experience Replay in Distributed Reinforcement Learning","paper":"/paper/recurrent-experience-replay-in-distributed","first_author":"Steven Kapturowski","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/recurrent-experience-replay-in-distributed"},"source":{"url":"https://openreview.net/forum?id=r1lyTjAqYX","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Offline Reinforcement Learning Methods","url":"/methods/category/offline-reinforcement-learning-methods","pwc_aliases":[]}],"n_papers_tagged":25,"archive_num_papers":25,"papers_newest_first":[{"paper":null,"title":"The R2D2 Deep Neural Network Series for Scalable Non-Cartesian Magnetic Resonance Imaging","date":"2025-03-12","arxiv_id":"2503.09559","n_code_links":0,"syntology":null},{"paper":null,"title":"Towards a robust R2D2 paradigm for radio-interferometric imaging: revisiting DNN training and architecture","date":"2025-03-04","arxiv_id":"2503.02554","n_code_links":0,"syntology":null},{"paper":null,"title":"S-R2D2: a spherical extension of the R2D2 deep neural network series paradigm for wide-field radio-interferometric imaging","date":"2025-03-03","arxiv_id":"2503.01462","n_code_links":0,"syntology":null},{"paper":null,"title":"R2D2: Remembering, Reflecting and Dynamic Decision Making for Web Agents","date":"2025-01-21","arxiv_id":"2501.12485","n_code_links":0,"syntology":null},{"paper":"/paper/a-deep-based-approach-for-multi-descriptor","title":"A Deep-Based Approach for Multi-Descriptor Feature Extraction: Applications on SAR Image Registration","date":"2024-11-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/targeted-latent-adversarial-training-improves","title":"Latent Adversarial Training Improves Robustness to Persistent Harmful Behaviors in LLMs","date":"2024-07-22","arxiv_id":"2407.15549","n_code_links":2,"syntology":{"ran":5,"of":11,"unverified":6,"pointer_only":0}},{"paper":"/paper/does-refusal-training-in-llms-generalize-to","title":"Does Refusal Training in LLMs Generalize to the Past Tense?","date":"2024-07-16","arxiv_id":"2407.11969","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":3}},{"paper":"/paper/simplifying-deep-temporal-difference-learning","title":"Simplifying Deep Temporal Difference Learning","date":"2024-07-05","arxiv_id":"2407.04811","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":0}},{"paper":"/paper/jailbreaking-leading-safety-aligned-llms-with","title":"Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks","date":"2024-04-02","arxiv_id":"2404.02151","n_code_links":1,"syntology":{"ran":8,"of":8,"unverified":0,"pointer_only":0}},{"paper":null,"title":"R2D2 image reconstruction with model uncertainty quantification in radio astronomy","date":"2024-03-26","arxiv_id":"2403.18052","n_code_links":0,"syntology":null},{"paper":null,"title":"Scalable Non-Cartesian Magnetic Resonance Imaging with R2D2","date":"2024-03-26","arxiv_id":"2403.17905","n_code_links":0,"syntology":null},{"paper":null,"title":"The R2D2 deep neural network series paradigm for fast precision imaging in radio astronomy","date":"2024-03-08","arxiv_id":"2403.05452","n_code_links":0,"syntology":null},{"paper":null,"title":"CLEANing Cygnus A deep and fast with R2D2","date":"2023-09-06","arxiv_id":"2309.03291","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-promise-and-limits-of-real-time","title":"Exploring the Promise and Limits of Real-Time Recurrent Learning","date":"2023-05-30","arxiv_id":"2305.19044","n_code_links":1,"syntology":null},{"paper":"/paper/hpointloc-point-based-indoor-place","title":"HPointLoc: Point-based Indoor Place Recognition using Synthetic RGB-D Images","date":"2022-12-30","arxiv_id":"2212.14649","n_code_links":1,"syntology":null},{"paper":"/paper/meta-referential-games-to-learn-compositional-1","title":"Meta-Referential Games to Learn Compositional Learning Behaviours","date":"2022-07-16","arxiv_id":"2207.08012","n_code_links":1,"syntology":null},{"paper":"/paper/r2d2-robust-data-to-text-with-replacement","title":"R2D2: Robust Data-to-Text with Replacement Detection","date":"2022-05-25","arxiv_id":"2205.12467","n_code_links":1,"syntology":null},{"paper":"/paper/zero-and-r2d2-a-large-scale-chinese-cross","title":"CCMB: A Large-scale Chinese Cross-modal Benchmark","date":"2022-05-08","arxiv_id":"2205.03860","n_code_links":1,"syntology":{"ran":0,"of":9,"unverified":9,"pointer_only":0}},{"paper":null,"title":"Semantic Exploration from Language Abstractions and Pretrained Representations","date":"2022-04-08","arxiv_id":"2204.05080","n_code_links":0,"syntology":null},{"paper":"/paper/fast-r2d2-a-pretrained-recursive-neural","title":"Fast-R2D2: A Pretrained Recursive Neural Network based on Pruned CKY for Grammar Induction and Text Representation","date":"2022-03-01","arxiv_id":"2203.00281","n_code_links":2,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":0}},{"paper":null,"title":"Retrieval-Augmented Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.08417","n_code_links":0,"syntology":null},{"paper":null,"title":"Statistically and Computationally Efficient Linear Meta-representation Learning","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/seed-rl-scalable-and-efficient-deep-rl-with-1","title":"SEED RL: Scalable and Efficient Deep-RL with Accelerated Central Inference","date":"2019-10-15","arxiv_id":"1910.06591","n_code_links":2,"syntology":{"ran":0,"of":6,"unverified":6,"pointer_only":0}},{"paper":null,"title":"R2D2: Reuse & Reduce via Dynamic Weight Diffusion for Training Efficient NLP Models","date":"2019-09-25","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-experience-replay-in-distributed","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"n_code_links":3,"syntology":null}],"papers_shown":25,"tasks":[{"task":"/task/image-reconstruction","name":"Image Reconstruction","papers":5},{"task":"/task/astronomy","name":"Astronomy","papers":4},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":4},{"task":"/task/retrieval","name":"Retrieval","papers":4},{"task":"/task/radio-interferometry","name":"Radio Interferometry","papers":3},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":3},{"task":"/task/image-retrieval","name":"Image Retrieval","papers":2},{"task":"/task/q-learning","name":"Q-Learning","papers":2},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":2},{"task":"/task/atari-games","name":"Atari Games","papers":1},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":1},{"task":"/task/constituency-grammar-induction","name":"Constituency Grammar Induction","papers":1},{"task":"/task/data-to-text-generation","name":"Data-to-Text Generation","papers":1},{"task":"/task/decision-making","name":"Decision Making","papers":1},{"task":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning","papers":1},{"task":"/task/denoising","name":"Denoising","papers":1},{"task":"/task/diagnostic","name":"Diagnostic","papers":1},{"task":"/task/entity-retrieval","name":"Entity Retrieval","papers":1},{"task":"/task/few-shot-learning","name":"Few-Shot Learning","papers":1},{"task":"/task/image-captioning","name":"Image Captioning","papers":1}],"tasks_shown":20,"n_tasks":47,"usage_by_year":[{"year":"2019","papers":3},{"year":"2021","papers":1},{"year":"2022","papers":7},{"year":"2023","papers":2},{"year":"2024","papers":8},{"year":"2025","papers":4}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/r2d2"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}