{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/ddpg/papers/3","list_of":"/method/ddpg","method":"DDPG","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,218],"of":218,"counts":{"archive_papers_tagged":218,"with_a_code_link":71,"where_syntology_ran_a_sample":14,"not_listed_spam_title":0,"listed":218,"listed_where_code_ran":14,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":13,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":13,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/ddpg","prev":"/method/ddpg/papers/2","next":null,"papers":[{"paper":"/paper/bipedal-walking-robot-using-deep","slug":"bipedal-walking-robot-using-deep","title":"Bipedal Walking Robot using Deep Deterministic Policy Gradient","date":"2018-07-16","arxiv_id":"1807.05924","n_code_links":3,"syntology":null},{"paper":null,"slug":"deterministic-policy-gradients-with-general","title":"Deterministic Policy Gradients With General State Transitions","date":"2018-07-10","arxiv_id":"1807.03708","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-explore-via-meta-policy-gradient","title":"Learning to Explore via Meta-Policy Gradient","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/stroke-based-character-reconstruction","slug":"stroke-based-character-reconstruction","title":"Stroke-based Character Reconstruction","date":"2018-06-23","arxiv_id":"1806.08990","n_code_links":1,"syntology":null},{"paper":"/paper/randomized-value-functions-via-multiplicative","slug":"randomized-value-functions-via-multiplicative","title":"Randomized Value Functions via Multiplicative Normalizing Flows","date":"2018-06-06","arxiv_id":"1806.02315","n_code_links":2,"syntology":null},{"paper":"/paper/advances-in-experience-replay","slug":"advances-in-experience-replay","title":"Advances in Experience Replay","date":"2018-05-15","arxiv_id":"1805.05536","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-explore-with-meta-policy-gradient","title":"Learning to Explore with Meta-Policy Gradient","date":"2018-03-13","arxiv_id":"1803.05044","n_code_links":0,"syntology":null},{"paper":"/paper/distributed-prioritized-experience-replay","slug":"distributed-prioritized-experience-replay","title":"Distributed Prioritized Experience Replay","date":"2018-03-02","arxiv_id":"1803.00933","n_code_links":15,"syntology":{"ran":9,"of":15,"n_ran_checked":9,"n_instrument":0,"unverified":6,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/gep-pg-decoupling-exploration-and","slug":"gep-pg-decoupling-exploration-and","title":"GEP-PG: Decoupling Exploration and Exploitation in Deep Reinforcement Learning Algorithms","date":"2018-02-14","arxiv_id":"1802.05054","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["flowersteam/geppg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"pretraining-deep-actor-critic-reinforcement","title":"Pretraining Deep Actor-Critic Reinforcement Learning Algorithms With Expert Demonstrations","date":"2018-01-31","arxiv_id":"1801.10459","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-run-with-actor-critic-ensemble","slug":"learning-to-run-with-actor-critic-ensemble","title":"Learning to Run with Actor-Critic Ensemble","date":"2017-12-25","arxiv_id":"1712.08987","n_code_links":2,"syntology":null},{"paper":"/paper/a-novel-ddpg-method-with-prioritized","slug":"a-novel-ddpg-method-with-prioritized","title":"A novel DDPG method with prioritized experience replay","date":"2017-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/leveraging-demonstrations-for-deep","slug":"leveraging-demonstrations-for-deep","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","date":"2017-07-27","arxiv_id":"1707.08817","n_code_links":4,"syntology":null},{"paper":null,"slug":"the-intentional-unintentional-agent-learning","title":"The Intentional Unintentional Agent: Learning to Solve Many Continuous Control Tasks Simultaneously","date":"2017-07-11","arxiv_id":"1707.03300","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-space-noise-for-exploration","slug":"parameter-space-noise-for-exploration","title":"Parameter Space Noise for Exploration","date":"2017-06-06","arxiv_id":"1706.01905","n_code_links":10,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"discrete-sequential-prediction-of-continuous","title":"Discrete Sequential Prediction of Continuous Actions for Deep RL","date":"2017-05-14","arxiv_id":"1705.05035","n_code_links":0,"syntology":null},{"paper":"/paper/actor-critic-versus-direct-policy-search-a","slug":"actor-critic-versus-direct-policy-search-a","title":"Actor-critic versus direct policy search: a comparison based on sample complexity","date":"2016-06-29","arxiv_id":"1606.09152","n_code_links":1,"syntology":null},{"paper":"/paper/continuous-control-with-deep-reinforcement","slug":"continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","n_code_links":161,"syntology":{"ran":163,"of":306,"n_ran_checked":152,"n_instrument":11,"unverified":143,"pointer_only":163,"phrase":"163 ran (of which 126 constructed an object rather than computing a result; 152 with no instrument failure: 3 honoured, 0 violated, 149 with no contract checked; 11 where Syntology's instrument failed) · 143 unverified","official":null}}],"record_sha256":"d80498168d7f25e81272299c72ead379dbf1fc705865e1822e1df06bd7c6faac","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}