{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/quantum-natural-policy-gradients-towards","title":"Quantum Natural Policy Gradients: Towards Sample-Efficient Reinforcement Learning","arxiv_id":"2304.13571","date":"2023-04-26","proceeding":null,"authors":["Nico Meyer","Daniel D. Scherer","Axel Plinge","Christopher Mutschler","Michael J. Hartmann"],"abstract":"Reinforcement learning is a growing field in AI with a lot of potential. Intelligent behavior is learned automatically through trial and error in interaction with the environment. However, this learning process is often costly. Using variational quantum circuits as function approximators potentially can reduce this cost. In order to implement this, we propose the quantum natural policy gradient (QNPG) algorithm -- a second-order gradient-based routine that takes advantage of an efficient approximation of the quantum Fisher information matrix. We experimentally demonstrate that QNPG outperforms first-order based training on Contextual Bandits environments regarding convergence speed and stability and moreover reduces the sample complexity. Furthermore, we provide evidence for the practical feasibility of our approach by training on a 12-qubit hardware device.","url_abs":"https://arxiv.org/abs/2304.13571v2","url_pdf":"https://arxiv.org/pdf/2304.13571v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"quantum-natural-policy-gradients-towards","repo_url":"https://github.com/nicomeyer96/quantum-natural-policy-gradients","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[{"task_slug":"multi-armed-bandits","task_name":"Multi-Armed Bandits"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[{"method_slug":"speed","method_name":"SPEED"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2304.13571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13571"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/nicomeyer96/quantum-natural-policy-gradients","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"unverified":4},"by_repo_kind":{"official":{"samples":4,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"f0603b7cf43f016e","entry":"compute_loss","repo":"nicomeyer96/quantum-natural-policy-gradients","repo_kind":"official","path":"utils.py","file_url":"https://github.com/nicomeyer96/quantum-natural-policy-gradients/blob/HEAD/utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f0603b7cf43f016e"}},{"code_sha256_prefix":"c3694885e1e558d1","entry":"postprocess_bitstrings","repo":"nicomeyer96/quantum-natural-policy-gradients","repo_kind":"official","path":"utils.py","file_url":"https://github.com/nicomeyer96/quantum-natural-policy-gradients/blob/HEAD/utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"c3694885e1e558d1"}},{"code_sha256_prefix":"720b07b98b308694","entry":"select_action","repo":"nicomeyer96/quantum-natural-policy-gradients","repo_kind":"official","path":"utils.py","file_url":"https://github.com/nicomeyer96/quantum-natural-policy-gradients/blob/HEAD/utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"720b07b98b308694"}},{"code_sha256_prefix":"d1b973464f5e3ccb","entry":"state_prep","repo":"nicomeyer96/quantum-natural-policy-gradients","repo_kind":"official","path":"environment.py","file_url":"https://github.com/nicomeyer96/quantum-natural-policy-gradients/blob/HEAD/environment.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d1b973464f5e3ccb"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}