{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/natural-language-reinforcement-learning-1","title":"Natural Language Reinforcement Learning","arxiv_id":"2411.14251","date":"2024-11-21","proceeding":null,"authors":["Xidong Feng","Bo Liu","Ziyu Wan","Haotian Fu","Girish A. Koushik","Zhiyuan Hu","Mengyue Yang","Ying Wen","Jun Wang"],"abstract":"Reinforcement Learning (RL) mathematically formulates decision-making with Markov Decision Process (MDP). With MDPs, researchers have achieved remarkable breakthroughs across various domains, including games, robotics, and language models. This paper seeks a new possibility, Natural Language Reinforcement Learning (NLRL), by extending traditional MDP to natural language-based representation space. Specifically, NLRL innovatively redefines RL principles, including task objectives, policy, value function, Bellman equation, and policy iteration, into their language counterparts. With recent advancements in large language models (LLMs), NLRL can be practically implemented to achieve RL-like policy and value improvement by either pure prompting or gradient-based training. Experiments over Maze, Breakthrough, and Tic-Tac-Toe games demonstrate the effectiveness, efficiency, and interpretability of the NLRL framework among diverse use cases.","url_abs":"https://arxiv.org/abs/2411.14251v2","url_pdf":"https://arxiv.org/pdf/2411.14251v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"natural-language-reinforcement-learning-1","repo_url":"https://github.com/waterhorse1/natural-language-rl","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"jax","reach":null}],"tasks":[{"task_slug":"decision-making","task_name":"Decision Making"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2411.14251","atlas_url":"https://app.syntology.ai/?focus=2411.14251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14251"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/waterhorse1/natural-language-rl","reach":null}],"summary":{"ran_honours":1,"ran_violates":2},"by_repo_kind":{"official":{"samples":3,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"a2863dd7dd994a0a","entry":"reward","repo":"waterhorse1/natural-language-rl","repo_kind":"official","path":"nlrl/evaluate.py","file_url":"https://github.com/waterhorse1/natural-language-rl/blob/HEAD/nlrl/evaluate.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a2863dd7dd994a0a"}},{"code_sha256_prefix":"de472f5f8aa647d0","entry":"tie_rate","repo":"waterhorse1/natural-language-rl","repo_kind":"official","path":"nlrl/evaluate.py","file_url":"https://github.com/waterhorse1/natural-language-rl/blob/HEAD/nlrl/evaluate.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"de472f5f8aa647d0"}},{"code_sha256_prefix":"e810010912b158b9","entry":"win_rate","repo":"waterhorse1/natural-language-rl","repo_kind":"official","path":"nlrl/evaluate.py","file_url":"https://github.com/waterhorse1/natural-language-rl/blob/HEAD/nlrl/evaluate.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e810010912b158b9"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}