{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/multi-objective-deep-reinforcement-learning","title":"Multi-Objective Deep Reinforcement Learning","arxiv_id":"1610.02707","date":"2016-10-09","proceeding":null,"authors":["Hossam Mossalam","Yannis M. Assael","Diederik M. Roijers","Shimon Whiteson"],"abstract":"We propose Deep Optimistic Linear Support Learning (DOL) to solve\nhigh-dimensional multi-objective decision problems where the relative\nimportances of the objectives are not known a priori. Using features from the\nhigh-dimensional inputs, DOL computes the convex coverage set containing all\npotential optimal solutions of the convex combinations of the objectives. To\nour knowledge, this is the first time that deep reinforcement learning has\nsucceeded in learning multi-objective policies. In addition, we provide a\ntestbed with two experiments to be used as a benchmark for deep multi-objective\nreinforcement learning.","url_abs":"http://arxiv.org/abs/1610.02707v1","url_pdf":"http://arxiv.org/pdf/1610.02707v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"multi-objective-deep-reinforcement-learning","repo_url":"https://github.com/hossam-mossalam/multi-objective-deep-rl","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"multi-objective-deep-reinforcement-learning","repo_url":"https://github.com/lucasalegre/morl-baselines","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"deep-reinforcement-learning","task_name":"Deep Reinforcement Learning"},{"task_slug":"multi-objective-reinforcement-learning","task_name":"Multi-Objective Reinforcement Learning"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/1610.02707","atlas_url":"https://app.syntology.ai/?focus=1610.02707","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}