{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/scheduled-policy-optimization-for-natural","title":"Scheduled Policy Optimization for Natural Language Communication with Intelligent Agents","arxiv_id":"1806.06187","date":"2018-06-16","proceeding":null,"authors":["Wenhan Xiong","Xiaoxiao Guo","Mo Yu","Shiyu Chang","Bo-Wen Zhou","William Yang Wang"],"abstract":"We investigate the task of learning to follow natural language instructions\nby jointly reasoning with visual observations and language inputs. In contrast\nto existing methods which start with learning from demonstrations (LfD) and\nthen use reinforcement learning (RL) to fine-tune the model parameters, we\npropose a novel policy optimization algorithm which dynamically schedules\ndemonstration learning and RL. The proposed training paradigm provides\nefficient exploration and better generalization beyond existing methods.\nComparing to existing ensemble models, the best single model based on our\nproposed method tremendously decreases the execution error by over 50% on a\nblock-world environment. To further illustrate the exploration strategy of our\nRL algorithm, We also include systematic studies on the evolution of policy\nentropy during training.","url_abs":"http://arxiv.org/abs/1806.06187v2","url_pdf":"http://arxiv.org/pdf/1806.06187v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"scheduled-policy-optimization-for-natural","repo_url":"https://github.com/xwhan/walk_the_blocks","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"scheduled-policy-optimization-for-natural","repo_url":"https://github.com/clic-lab/ciff","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"GPL-3.0"}},{"paper_slug":"scheduled-policy-optimization-for-natural","repo_url":"https://github.com/lil-lab/ciff","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"GPL-3.0"}}],"tasks":[{"task_slug":"efficient-exploration","task_name":"Efficient Exploration"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}