{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/reinforcement-learning-with-random-delays-1","title":"Reinforcement Learning with Random Delays","arxiv_id":"2010.02966","date":"2020-10-06","proceeding":"ICLR 2021 1","authors":["Simon Ramstedt","Yann Bouteiller","Giovanni Beltrame","Christopher Pal","Jonathan Binas"],"abstract":"Action and observation delays commonly occur in many Reinforcement Learning applications, such as remote control scenarios. We study the anatomy of randomly delayed environments, and show that partially resampling trajectory fragments in hindsight allows for off-policy multi-step value estimation. We apply this principle to derive Delay-Correcting Actor-Critic (DCAC), an algorithm based on Soft Actor-Critic with significantly better performance in environments with delays. This is shown theoretically and also demonstrated practically on a delay-augmented version of the MuJoCo continuous control benchmark.","url_abs":"https://arxiv.org/abs/2010.02966v3","url_pdf":"https://arxiv.org/pdf/2010.02966v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"reinforcement-learning-with-random-delays-1","repo_url":"https://github.com/rmst/rlrd","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"reinforcement-learning-with-random-delays-1","repo_url":"https://github.com/cav-research-lab/predictive-model-delay-correction","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"reinforcement-learning-with-random-delays-1","repo_url":"https://github.com/yannbouteiller/rtgym","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null}],"tasks":[{"task_slug":"anatomy","task_name":"Anatomy"},{"task_slug":"continuous-control","task_name":"Continuous Control"},{"task_slug":"mujoco","task_name":"MuJoCo"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"continuous-control","task_name":"continuous-control"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2010.02966","atlas_url":"https://app.syntology.ai/?focus=2010.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02966"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/rmst/rlrd","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/yannbouteiller/rtgym","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cav-research-lab/predictive-model-delay-correction","reach":null}],"summary":{"ran":2,"unverified":7},"by_repo_kind":{"official":{"samples":4,"ran":0,"repositories":1},"listed":{"samples":5,"ran":2,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"cb6a59084fc801ca","entry":"Benchmark","repo":"yannbouteiller/rtgym","repo_kind":"listed","path":"rtgym/envs/real_time_env.py","file_url":"https://github.com/yannbouteiller/rtgym/blob/HEAD/rtgym/envs/real_time_env.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"cb6a59084fc801ca"}},{"code_sha256_prefix":"8297c2568af64411","entry":"DCNN","repo":"cav-research-lab/predictive-model-delay-correction","repo_kind":"listed","path":"delay_correcting_nn.py","file_url":"https://github.com/cav-research-lab/predictive-model-delay-correction/blob/HEAD/delay_correcting_nn.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8297c2568af64411"}},{"code_sha256_prefix":"c72b16e1588e157e","entry":"RealTimeEnv","repo":"yannbouteiller/rtgym","repo_kind":"listed","path":"rtgym/envs/real_time_env.py","file_url":"https://github.com/yannbouteiller/rtgym/blob/HEAD/rtgym/envs/real_time_env.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c72b16e1588e157e"}},{"code_sha256_prefix":"2b682fa249706cb8","entry":"RealTimeGymInterface","repo":"yannbouteiller/rtgym","repo_kind":"listed","path":"rtgym/envs/real_time_env.py","file_url":"https://github.com/yannbouteiller/rtgym/blob/HEAD/rtgym/envs/real_time_env.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"2b682fa249706cb8"}},{"code_sha256_prefix":"e6fd69a2f1d0511a","entry":"TraceBenchmark","repo":"yannbouteiller/rtgym","repo_kind":"listed","path":"rtgym/envs/real_time_env.py","file_url":"https://github.com/yannbouteiller/rtgym/blob/HEAD/rtgym/envs/real_time_env.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e6fd69a2f1d0511a"}},{"code_sha256_prefix":"f3ed23b35cd6dba4","entry":"copy_shared","repo":"rmst/rlrd","repo_kind":"official","path":"rlrd/nn.py","file_url":"https://github.com/rmst/rlrd/blob/HEAD/rlrd/nn.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f3ed23b35cd6dba4"}},{"code_sha256_prefix":"9bf0c974e2a7db1f","entry":"detach","repo":"rmst/rlrd","repo_kind":"official","path":"rlrd/nn.py","file_url":"https://github.com/rmst/rlrd/blob/HEAD/rlrd/nn.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"9bf0c974e2a7db1f"}},{"code_sha256_prefix":"f579aaa1966a6d09","entry":"get_env_state","repo":"rmst/rlrd","repo_kind":"official","path":"rlrd/batch_env.py","file_url":"https://github.com/rmst/rlrd/blob/HEAD/rlrd/batch_env.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f579aaa1966a6d09"}},{"code_sha256_prefix":"064f68f60a5e4279","entry":"no_grad","repo":"rmst/rlrd","repo_kind":"official","path":"rlrd/nn.py","file_url":"https://github.com/rmst/rlrd/blob/HEAD/rlrd/nn.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"064f68f60a5e4279"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}