{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/smx-sequential-monte-carlo-planning-for","title":"SPO: Sequential Monte Carlo Policy Optimisation","arxiv_id":"2402.07963","date":"2024-02-12","proceeding":null,"authors":["Matthew V Macfarlane","Edan Toledo","Donal Byrne","Paul Duckworth","Alexandre Laterre"],"abstract":"Leveraging planning during learning and decision-making is central to the long-term development of intelligent agents. Recent works have successfully combined tree-based search methods and self-play learning mechanisms to this end. However, these methods typically face scaling challenges due to the sequential nature of their search. While practical engineering solutions can partly overcome this, they often result in a negative impact on performance. In this paper, we introduce SPO: Sequential Monte Carlo Policy Optimisation, a model-based reinforcement learning algorithm grounded within the Expectation Maximisation (EM) framework. We show that SPO provides robust policy improvement and efficient scaling properties. The sample-based search makes it directly applicable to both discrete and continuous action spaces without modifications. We demonstrate statistically significant improvements in performance relative to model-free and model-based baselines across both continuous and discrete environments. Furthermore, the parallel nature of SPO's search enables effective utilisation of hardware accelerators, yielding favourable scaling laws.","url_abs":"https://arxiv.org/abs/2402.07963v3","url_pdf":"https://arxiv.org/pdf/2402.07963v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"smx-sequential-monte-carlo-planning-for","repo_url":"https://github.com/edantoledo/stoix","is_official":0,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"jax","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"decision-making","task_name":"Decision Making"},{"task_slug":"model-based-reinforcement-learning","task_name":"Model-based Reinforcement Learning"},{"task_slug":"self-learning","task_name":"Self-Learning"}],"methods":[{"method_slug":"alphazero","method_name":"AlphaZero"},{"method_slug":"self-learning","method_name":"Self-Learning"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2402.07963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07963"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/edantoledo/stoix","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"unverified":5},"by_repo_kind":{"named_in_paper":{"samples":5,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"33e63dbaf7e1a19b","entry":"chained_torsos","repo":"edantoledo/stoix","repo_kind":"named_in_paper","path":"stoix/networks/base.py","file_url":"https://github.com/edantoledo/stoix/blob/HEAD/stoix/networks/base.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"33e63dbaf7e1a19b"}},{"code_sha256_prefix":"5b5a4dd5ab4649c5","entry":"clip_to_spec","repo":"edantoledo/stoix","repo_kind":"named_in_paper","path":"stoix/networks/postprocessors.py","file_url":"https://github.com/edantoledo/stoix/blob/HEAD/stoix/networks/postprocessors.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"5b5a4dd5ab4649c5"}},{"code_sha256_prefix":"de1e17a2985419aa","entry":"make_downsampling_layer","repo":"edantoledo/stoix","repo_kind":"named_in_paper","path":"stoix/networks/resnet.py","file_url":"https://github.com/edantoledo/stoix/blob/HEAD/stoix/networks/resnet.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"de1e17a2985419aa"}},{"code_sha256_prefix":"2967a9b5286c6007","entry":"rescale_to_spec","repo":"edantoledo/stoix","repo_kind":"named_in_paper","path":"stoix/networks/postprocessors.py","file_url":"https://github.com/edantoledo/stoix/blob/HEAD/stoix/networks/postprocessors.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"2967a9b5286c6007"}},{"code_sha256_prefix":"2c00c132a3cc27e0","entry":"tanh_to_spec","repo":"edantoledo/stoix","repo_kind":"named_in_paper","path":"stoix/networks/postprocessors.py","file_url":"https://github.com/edantoledo/stoix/blob/HEAD/stoix/networks/postprocessors.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"2c00c132a3cc27e0"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}