{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/deep-reinforcement-learning-in-a-handful-of","title":"Deep Reinforcement Learning in a Handful of Trials using Probabilistic Dynamics Models","arxiv_id":"1805.12114","date":"2018-05-30","proceeding":"NeurIPS 2018 12","authors":["Kurtland Chua","Roberto Calandra","Rowan Mcallister","Sergey Levine"],"abstract":"Model-based reinforcement learning (RL) algorithms can attain excellent\nsample efficiency, but often lag behind the best model-free algorithms in terms\nof asymptotic performance. This is especially true with high-capacity\nparametric function approximators, such as deep networks. In this paper, we\nstudy how to bridge this gap, by employing uncertainty-aware dynamics models.\nWe propose a new algorithm called probabilistic ensembles with trajectory\nsampling (PETS) that combines uncertainty-aware deep network dynamics models\nwith sampling-based uncertainty propagation. Our comparison to state-of-the-art\nmodel-based and model-free deep RL algorithms shows that our approach matches\nthe asymptotic performance of model-free algorithms on several challenging\nbenchmark tasks, while requiring significantly fewer samples (e.g., 8 and 125\ntimes fewer samples than Soft Actor Critic and Proximal Policy Optimization\nrespectively on the half-cheetah task).","url_abs":"http://arxiv.org/abs/1805.12114v2","url_pdf":"http://arxiv.org/pdf/1805.12114v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/kchua/handful-of-trials","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/ByMic/PETS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/Shunichi09/PythonLinearNonlinearControl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/facebookresearch/mbrl-lib","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/github-jnauta/pytorch-pne","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/jingwu6/handful-of-trials-in-pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/natolambert/dynamicslearn","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/quanvuong/handful-of-trials-pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"deep-reinforcement-learning-in-a-handful-of","repo_url":"https://github.com/sradicwebster/mbrl-lib","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"deep-reinforcement-learning","task_name":"Deep Reinforcement Learning"},{"task_slug":"model-based-reinforcement-learning","task_name":"Model-based Reinforcement Learning"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"experience-replay","method_name":"Experience Replay"},{"method_slug":"relu","method_name":"ReLU"},{"method_slug":"soft-actor-critic","method_name":"Soft Actor Critic"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1805.12114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.12114"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/kchua/handful-of-trials","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/github-jnauta/pytorch-pne","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/sradicwebster/mbrl-lib","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/natolambert/dynamicslearn","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/facebookresearch/mbrl-lib","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jingwu6/handful-of-trials-in-pytorch","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Shunichi09/PythonLinearNonlinearControl","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ByMic/PETS","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/quanvuong/handful-of-trials-pytorch","reach":null}],"summary":{"ran":6,"ran_draft_wrong":2,"ran_fixture":3,"ran_honours":1,"unverified":7},"by_repo_kind":{"listed":{"samples":19,"ran":12,"repositories":4}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":17,"samples":[{"code_sha256_prefix":"f13fc61d4b56b01e","entry":"BNN","repo":"jingwu6/handful-of-trials-in-pytorch","repo_kind":"listed","path":"modeling/models/BNN.py","file_url":"https://github.com/jingwu6/handful-of-trials-in-pytorch/blob/HEAD/modeling/models/BNN.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f13fc61d4b56b01e"}},{"code_sha256_prefix":"eeb421621d540f5f","entry":"CEMOptimizer","repo":"quanvuong/handful-of-trials-pytorch","repo_kind":"listed","path":"MPC.py","file_url":"https://github.com/quanvuong/handful-of-trials-pytorch/blob/HEAD/MPC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"eeb421621d540f5f"}},{"code_sha256_prefix":"e4ea9ddc4d8483c2","entry":"Controller","repo":"quanvuong/handful-of-trials-pytorch","repo_kind":"listed","path":"MPC.py","file_url":"https://github.com/quanvuong/handful-of-trials-pytorch/blob/HEAD/MPC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e4ea9ddc4d8483c2"}},{"code_sha256_prefix":"5ca4e528d445b8bb","entry":"DNN","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5ca4e528d445b8bb"}},{"code_sha256_prefix":"5938fc919ec8c9ab","entry":"Optimizer","repo":"quanvuong/handful-of-trials-pytorch","repo_kind":"listed","path":"MPC.py","file_url":"https://github.com/quanvuong/handful-of-trials-pytorch/blob/HEAD/MPC.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5938fc919ec8c9ab"}},{"code_sha256_prefix":"a9c1528dfcc4652a","entry":"PNN","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a9c1528dfcc4652a"}},{"code_sha256_prefix":"4eb5226056a1f70f","entry":"get_required_argument","repo":"quanvuong/handful-of-trials-pytorch","repo_kind":"listed","path":"MPC.py","file_url":"https://github.com/quanvuong/handful-of-trials-pytorch/blob/HEAD/MPC.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"4eb5226056a1f70f"}},{"code_sha256_prefix":"8a600321cfd91f4d","entry":"normalize_deltas","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8a600321cfd91f4d"}},{"code_sha256_prefix":"f1b4f8263ddf0967","entry":"normalize_obs","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f1b4f8263ddf0967"}},{"code_sha256_prefix":"14a3c589a22e7e32","entry":"shuffle_rows","repo":"quanvuong/handful-of-trials-pytorch","repo_kind":"listed","path":"MPC.py","file_url":"https://github.com/quanvuong/handful-of-trials-pytorch/blob/HEAD/MPC.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"14a3c589a22e7e32"}},{"code_sha256_prefix":"35f8aeec75f44df1","entry":"silu","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"35f8aeec75f44df1"}},{"code_sha256_prefix":"09253c3b2b54a147","entry":"unnormalize_deltas","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"09253c3b2b54a147"}},{"code_sha256_prefix":"1c0e747272c22445","entry":"Ensemble","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"1c0e747272c22445"}},{"code_sha256_prefix":"d45086d0d8ac50af","entry":"EnsembleNN","repo":"natolambert/dynamicslearn","repo_kind":"listed","path":"learn/models/model_ensemble_nn.py","file_url":"https://github.com/natolambert/dynamicslearn/blob/HEAD/learn/models/model_ensemble_nn.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d45086d0d8ac50af"}},{"code_sha256_prefix":"d696ce3649bc23c0","entry":"GeneralNN","repo":"natolambert/dynamicslearn","repo_kind":"listed","path":"learn/models/model_ensemble_nn.py","file_url":"https://github.com/natolambert/dynamicslearn/blob/HEAD/learn/models/model_ensemble_nn.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d696ce3649bc23c0"}},{"code_sha256_prefix":"88a7bccc5d349061","entry":"MPC","repo":"quanvuong/handful-of-trials-pytorch","repo_kind":"listed","path":"MPC.py","file_url":"https://github.com/quanvuong/handful-of-trials-pytorch/blob/HEAD/MPC.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"88a7bccc5d349061"}},{"code_sha256_prefix":"139faffd313f0eb3","entry":"get_affine_params","repo":"jingwu6/handful-of-trials-in-pytorch","repo_kind":"listed","path":"modeling/models/BNN.py","file_url":"https://github.com/jingwu6/handful-of-trials-in-pytorch/blob/HEAD/modeling/models/BNN.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"139faffd313f0eb3"}},{"code_sha256_prefix":"48fe26920f96d27b","entry":"init_weights","repo":"ByMic/PETS","repo_kind":"listed","path":"utils/ensemble.py","file_url":"https://github.com/ByMic/PETS/blob/HEAD/utils/ensemble.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"48fe26920f96d27b"}},{"code_sha256_prefix":"24b8d2393ffaa19e","entry":"truncated_normal","repo":"jingwu6/handful-of-trials-in-pytorch","repo_kind":"listed","path":"modeling/models/BNN.py","file_url":"https://github.com/jingwu6/handful-of-trials-in-pytorch/blob/HEAD/modeling/models/BNN.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"24b8d2393ffaa19e"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}