{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/ddpgagent","entry":"DDPGAgent","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":5,"n_papers_ran":2,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":15,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":15,"n_places_pointer_only":8,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":4,"unverified":11},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2604.19146","paper":"/paper/arxiv-2604-19146","title":"RL-ABC: Reinforcement Learning for Accelerator Beamline Control","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Anwar9Ibrahim/RL-ABC","path":"rl_framework/Agents/DDPG.py","file_url":"https://github.com/Anwar9Ibrahim/RL-ABC/blob/HEAD/rl_framework/Agents/DDPG.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea447dde584febcb","mcp_get_code":{"code_sha256":"ea447dde584febcb"}},{"arxiv_id":"2406.16255","paper":"/paper/uncertainty-aware-reward-free-exploration","title":"Uncertainty-Aware Reward-Free Exploration with General Function Approximation","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uclaml/gfa-rfe","path":"agent/dsquare.py","file_url":"https://github.com/uclaml/gfa-rfe/blob/HEAD/agent/dsquare.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4c262c61dad0f383","mcp_get_code":{"code_sha256":"4c262c61dad0f383"}},{"arxiv_id":"2103.04551","paper":"/paper/behavior-from-the-void-unsupervised-active","title":"Behavior From the Void: Unsupervised Active Pre-Training","date":"2021-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rll-research/url_benchmark","path":"agent/icm_apt.py","file_url":"https://github.com/rll-research/url_benchmark/blob/HEAD/agent/icm_apt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"875a14ea71682fe8","mcp_get_code":{"code_sha256":"875a14ea71682fe8"}},{"arxiv_id":"1706.02275","paper":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"biemann/Collaboration-and-Competition","path":"maddpg.py","file_url":"https://github.com/biemann/Collaboration-and-Competition/blob/HEAD/maddpg.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bbf8b2fce98ed3c3","mcp_get_code":{"code_sha256":"bbf8b2fce98ed3c3"}},{"arxiv_id":"1706.02275","paper":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thechrisyoon08/marl","path":"MADDPG/maddpg.py","file_url":"https://github.com/thechrisyoon08/marl/blob/HEAD/MADDPG/maddpg.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4471e9be4006d33f","mcp_get_code":{"code_sha256":"4471e9be4006d33f"}},{"arxiv_id":"1706.02275","paper":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yutongamber/MADDPG","path":"maddpg-pytorch/algorithms/maddpg.py","file_url":"https://github.com/Yutongamber/MADDPG/blob/HEAD/maddpg-pytorch/algorithms/maddpg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"23ccb70277ce98c1","mcp_get_code":{"code_sha256":"23ccb70277ce98c1"}},{"arxiv_id":"1706.02275","paper":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mauricemager/multiagent-robot","path":"algorithms/maddpg.py","file_url":"https://github.com/mauricemager/multiagent-robot/blob/HEAD/algorithms/maddpg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"798b209df2b695ff","mcp_get_code":{"code_sha256":"798b209df2b695ff"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dpoulopoulos/drl_collaborate_compete","path":"ddpg_agent.py","file_url":"https://github.com/dpoulopoulos/drl_collaborate_compete/blob/HEAD/ddpg_agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"037ce510df007080","mcp_get_code":{"code_sha256":"037ce510df007080"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HJDQN/HJQ","path":"algorithms/ddpg/ddpg.py","file_url":"https://github.com/HJDQN/HJQ/blob/HEAD/algorithms/ddpg/ddpg.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b51369036ab5a352","mcp_get_code":{"code_sha256":"b51369036ab5a352"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"J93T/TP4-DDPG","path":"ddpg.py","file_url":"https://github.com/J93T/TP4-DDPG/blob/HEAD/ddpg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d96fe6b321763c14","mcp_get_code":{"code_sha256":"d96fe6b321763c14"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"InSpaceAI/RL-Zoo","path":"DDPG.py","file_url":"https://github.com/InSpaceAI/RL-Zoo/blob/HEAD/DDPG.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4b11a27451c374cd","mcp_get_code":{"code_sha256":"4b11a27451c374cd"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hemilpanchiwala/Hindsight-Experience-Replay","path":"ddpg_with_her/DDPGAgent.py","file_url":"https://github.com/hemilpanchiwala/Hindsight-Experience-Replay/blob/HEAD/ddpg_with_her/DDPGAgent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"76a41f814d76f5c0","mcp_get_code":{"code_sha256":"76a41f814d76f5c0"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"denizmguen/IANNWTF2019-Project","path":"ddpg.py","file_url":"https://github.com/denizmguen/IANNWTF2019-Project/blob/HEAD/ddpg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5b513793c76f96d8","mcp_get_code":{"code_sha256":"5b513793c76f96d8"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"soumik12345/DDPG","path":"src/agents.py","file_url":"https://github.com/soumik12345/DDPG/blob/HEAD/src/agents.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fb2a2e6b4cc0d551","mcp_get_code":{"code_sha256":"fb2a2e6b4cc0d551"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KelvinYang0320/deepbots-panda","path":"Panda_RL/controllers/robot_supervisor_manager/agent/ddpg.py","file_url":"https://github.com/KelvinYang0320/deepbots-panda/blob/HEAD/Panda_RL/controllers/robot_supervisor_manager/agent/ddpg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c0e725d5bf9fb153","mcp_get_code":{"code_sha256":"c0e725d5bf9fb153"}}]}