{"url":"/task/multi-objective-reinforcement-learning","name":"Multi-Objective Reinforcement Learning","slug":"multi-objective-reinforcement-learning","description_markdown":null,"categories":[{"name":"Computer Code","url":"/area/computer-code"},{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Knowledge Base","url":"/area/knowledge-base"},{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":143,"papers_with_code":55,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[{"url":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":55,"tagged_in_all":143,"items":[{"url":"/paper/optimization-of-molecules-via-deep","title":"Optimization of Molecules via Deep Reinforcement Learning","date":"2018-10-19","arxiv_id":"1810.08678","repositories_listed":7,"syntology":null},{"url":"/paper/a-generalized-algorithm-for-multi-objective","title":"A Generalized Algorithm for Multi-Objective Reinforcement Learning and Policy Adaptation","date":"2019-08-21","arxiv_id":"1908.08342","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/sample-efficient-multi-objective-learning-via","title":"Sample-Efficient Multi-Objective Learning via Generalized Policy Improvement Prioritization","date":"2023-01-18","arxiv_id":"2301.07784","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/dynamic-weights-in-multi-objective-deep","title":"Dynamic Weights in Multi-Objective Deep Reinforcement Learning","date":"2018-09-20","arxiv_id":"1809.07803","repositories_listed":3,"syntology":null},{"url":"/paper/a-toolkit-for-reliable-benchmarking-and","title":"A Toolkit for Reliable Benchmarking and Research in Multi-Objective Reinforcement Learning","date":"2023-09-26","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/mo-gym-a-library-of-multi-objective","title":"MO-Gym: A Library of Multi-Objective Reinforcement Learning Environments","date":"2022-11-30","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/anchor-changing-regularized-natural-policy","title":"Anchor-Changing Regularized Natural Policy Gradient for Multi-Objective Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05357","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/prediction-guided-multi-objective","title":"Prediction-Guided Multi-Objective Reinforcement Learning for Continuous Robot Control","date":"2020-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/multi-objective-deep-reinforcement-learning","title":"Multi-Objective Deep Reinforcement Learning","date":"2016-10-09","arxiv_id":"1610.02707","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-moeas-for-solving-continuous","title":"Benchmarking MOEAs for solving continuous multi-objective RL problems","date":"2025-05-19","arxiv_id":"2505.13726","repositories_listed":1,"syntology":null},{"url":"/paper/active-sampling-for-mri-based-sequential","title":"Active Sampling for MRI-based Sequential Decision Making","date":"2025-05-07","arxiv_id":"2505.04586","repositories_listed":1,"syntology":null},{"url":"/paper/emorl-ensemble-multi-objective-reinforcement","title":"EMORL: Ensemble Multi-Objective Reinforcement Learning for Efficient and Flexible LLM Fine-Tuning","date":"2025-05-05","arxiv_id":"2505.02579","repositories_listed":1,"syntology":null},{"url":"/paper/on-generalization-across-environments-in","title":"On Generalization Across Environments In Multi-Objective Reinforcement Learning","date":"2025-03-02","arxiv_id":"2503.00799","repositories_listed":1,"syntology":null},{"url":"/paper/mol-moe-training-preference-guided-routers","title":"Mol-MoE: Training Preference-Guided Routers for Molecule Generation","date":"2025-02-08","arxiv_id":"2502.05633","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/multi-objective-reinforcement-learning-for-2","title":"Multi-Objective Reinforcement Learning for Power Grid Topology Control","date":"2025-01-27","arxiv_id":"2502.00040","repositories_listed":1,"syntology":null},{"url":"/paper/psmgd-periodic-stochastic-multi-gradient","title":"PSMGD: Periodic Stochastic Multi-Gradient Descent for Fast Multi-Objective Optimization","date":"2024-12-14","arxiv_id":"2412.10961","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-multi-objective-reinforcement","title":"Scalable Multi-Objective Reinforcement Learning with Fairness Guarantees using Lorenz Dominance","date":"2024-11-27","arxiv_id":"2411.18195","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-trade-offs-policy-summarization","title":"Navigating Trade-offs: Policy Summarization for Multi-Objective Reinforcement Learning","date":"2024-11-07","arxiv_id":"2411.04784","repositories_listed":1,"syntology":null},{"url":"/paper/c-morl-multi-objective-reinforcement-learning","title":"C-MORL: Multi-Objective Reinforcement Learning through Efficient Discovery of Pareto Front","date":"2024-10-03","arxiv_id":"2410.02236","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":3}},{"url":"/paper/inferring-preferences-from-demonstrations-in-2","title":"Inferring Preferences from Demonstrations in Multi-objective Reinforcement Learning","date":"2024-09-30","arxiv_id":"2409.20258","repositories_listed":1,"syntology":null},{"url":"/paper/stage-wise-reward-shaping-for-acrobatic","title":"Stage-Wise Reward Shaping for Acrobatic Robots: A Constrained Multi-Objective Reinforcement Learning Approach","date":"2024-09-24","arxiv_id":"2409.15755","repositories_listed":1,"syntology":null},{"url":"/paper/learning-pareto-set-for-multi-objective","title":"Learning Pareto Set for Multi-Objective Continuous Robot Control","date":"2024-06-27","arxiv_id":"2406.18924","repositories_listed":1,"syntology":null},{"url":"/paper/the-max-min-formulation-of-multi-objective","title":"The Max-Min Formulation of Multi-Objective Reinforcement Learning: From Theory to a Model-Free Algorithm","date":"2024-06-12","arxiv_id":"2406.07826","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/multi-objective-reinforcement-learning-from","title":"Multi-objective Reinforcement learning from AI Feedback","date":"2024-06-11","arxiv_id":"2406.07295","repositories_listed":1,"syntology":null},{"url":"/paper/deep-multi-objective-reinforcement-learning","title":"Deep Multi-Objective Reinforcement Learning for Utility-Based Infrastructural Maintenance Optimization","date":"2024-06-10","arxiv_id":"2406.06184","repositories_listed":1,"syntology":null},{"url":"/paper/spatio-temporal-early-prediction-based-on","title":"STEMO: Early Spatio-temporal Forecasting with Multi-Objective Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.04035","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/safe-and-balanced-a-framework-for-constrained","title":"Safe and Balanced: A Framework for Constrained Multi-Objective Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16390","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/ucb-driven-utility-function-search-for-multi","title":"UCB-driven Utility Function Search for Multi-objective Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.00410","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-dynamic-multi","title":"Dynamic Multi-Reward Weighting for Multi-Style Controllable Generation","date":"2024-02-21","arxiv_id":"2402.14146","repositories_listed":1,"syntology":null},{"url":"/paper/divide-and-conquer-provably-unveiling-the","title":"Divide and Conquer: Provably Unveiling the Pareto Front with Multi-Objective Reinforcement Learning","date":"2024-02-11","arxiv_id":"2402.07182","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_unverified":1,"n_pointer_only":17}}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}