{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multi-objective-reinforcement-learning/papers/2","list_of":"/task/multi-objective-reinforcement-learning","task":"Multi-Objective Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,143],"of":143,"counts":{"archive_papers_tagged":143,"with_a_code_link":55,"where_syntology_ran_a_sample":19,"not_listed_spam_title":0,"listed":143,"listed_where_code_ran":19,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":19,"every_run_a_failure_of_syntologys_instrument":0,"listed_with_a_run_with_no_instrument_failure":19,"listed_every_run_a_failure_of_syntologys_instrument":0,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multi-objective-reinforcement-learning","prev":"/task/multi-objective-reinforcement-learning","next":null,"papers":[{"url":null,"slug":"on-the-expressivity-of-objective","title":"On The Expressivity of Objective-Specification Formalisms in Reinforcement Learning","date":"2023-10-18","arxiv_id":"2310.11840","repositories_listed":0,"syntology":null},{"url":null,"slug":"urban-drone-navigation-autoencoder-learning","title":"Urban Drone Navigation: Autoencoder Learning Fusion for Aerodynamics","date":"2023-10-13","arxiv_id":"2310.08830","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-robust-policy-bootstrapping-algorithm-for","title":"A Robust Policy Bootstrapping Algorithm for Multi-objective Reinforcement Learning in Non-stationary Environments","date":"2023-08-18","arxiv_id":"2308.09734","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-hierarchical-policy","title":"Intrinsically Motivated Hierarchical Policy Learning in Multi-objective Markov Decision Processes","date":"2023-08-18","arxiv_id":"2308.09733","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multiobjective-reinforcement-learning","title":"A Multiobjective Reinforcement Learning Framework for Microgrid Energy Management","date":"2023-07-17","arxiv_id":"2307.08692","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-objective-reinforcement","title":"Cooperative Multi-Objective Reinforcement Learning for Traffic Signal Control and Carbon Emission Reduction","date":"2023-06-16","arxiv_id":"2306.09662","repositories_listed":0,"syntology":null},{"url":null,"slug":"inferring-preferences-from-demonstrations-in","title":"Inferring Preferences from Demonstrations in Multi-objective Reinforcement Learning: A Dynamic Weight-based Approach","date":"2023-04-27","arxiv_id":"2304.14115","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-conditioned-policy-gradient-for-multi","title":"Latent-Conditioned Policy Gradient for Multi-Objective Deep Reinforcement Learning","date":"2023-03-15","arxiv_id":"2303.08909","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scale-independent-multi-objective","title":"A Scale-Independent Multi-Objective Reinforcement Learning with Convergence Analysis","date":"2023-02-08","arxiv_id":"2302.04179","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-lagrangian-chaos-with-multi-objective","title":"Taming Lagrangian Chaos with Multi-Objective Reinforcement Learning","date":"2022-12-19","arxiv_id":"2212.09612","repositories_listed":0,"syntology":null},{"url":null,"slug":"elixir-a-system-to-enhance-data-quality-for","title":"Elixir: A system to enhance data quality for multiple analytics on a video stream","date":"2022-12-08","arxiv_id":"2212.04061","repositories_listed":0,"syntology":null},{"url":null,"slug":"monte-carlo-tree-search-algorithms-for-risk","title":"Monte Carlo Tree Search Algorithms for Risk-Aware and Multi-Objective Reinforcement Learning","date":"2022-11-23","arxiv_id":"2211.13032","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-the-issue-of-stochastic","title":"Addressing the issue of stochastic environments and local decision-making in multi-objective reinforcement learning","date":"2022-11-16","arxiv_id":"2211.08669","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-coordination-graphs-for-the","title":"Multi-Objective Coordination Graphs for the Expected Scalarised Returns with Generative Flow Models","date":"2022-07-01","arxiv_id":"2207.00368","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-pareto-front-of-multi-objective","title":"Exploring the Pareto front of multi-objective COVID-19 mitigation policies using reinforcement learning","date":"2022-04-11","arxiv_id":"2204.05027","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-staged-rollout-with-reinforcement","title":"Automating Staged Rollout with Reinforcement Learning","date":"2022-04-01","arxiv_id":"2204.02189","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-multi-objective-reinforcement","title":"Evolutionary Multi-Objective Reinforcement Learning Based Trajectory Control and Task Offloading in UAV-Assisted Mobile Edge Computing","date":"2022-02-24","arxiv_id":"2202.12028","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviour-diverse-automatic-penetration","title":"Behaviour-Diverse Automatic Penetration Testing: A Curiosity-Driven Multi-Objective Deep Reinforcement Learning Approach","date":"2022-02-22","arxiv_id":"2202.10630","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-pareto-efficient-fairness-utility","title":"Toward Pareto Efficient Fairness-Utility Trade-off inRecommendation through Reinforcement Learning","date":"2022-01-01","arxiv_id":"2201.00140","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-constrained-multi-objective","title":"Offline Constrained Multi-Objective Reinforcement Learning via Pessimistic Dual Value Iteration","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"choosing-the-best-of-both-worlds-diverse-and","title":"Choosing the Best of Both Worlds: Diverse and Novel Recommendations through Multi-Objective Reinforcement Learning","date":"2021-10-28","arxiv_id":"2110.15097","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-in-urban-environments-amongst","title":"Navigation In Urban Environments Amongst Pedestrians Using Multi-Objective Deep Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05205","repositories_listed":0,"syntology":null},{"url":null,"slug":"pareto-policy-adaptation","title":"Pareto Policy Adaptation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"expected-scalarised-returns-dominance-a-new","title":"Expected Scalarised Returns Dominance: A New Solution Concept for Multi-Objective Decision Making","date":"2021-06-02","arxiv_id":"2106.01048","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-optimization-of-multi-objective","title":"Joint Optimization of Multi-Objective Reinforcement Learning with Policy Gradient Based Algorithm","date":"2021-05-28","arxiv_id":"2105.14125","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-predictive-control-and-reinforcement","title":"Model-predictive control and reinforcement learning in multi-energy system case studies","date":"2021-04-20","arxiv_id":"2104.09785","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-based-1","title":"Multi-Objective Reinforcement Learning based Multi-Microgrid System Optimisation Problem","date":"2021-03-10","arxiv_id":"2103.06380","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-algorithms-for-multi","title":"Provably Efficient Algorithms for Multi-Objective Competitive RL","date":"2021-02-05","arxiv_id":"2102.03192","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-aware-and-multi-objective-decision","title":"Risk Aware and Multi-Objective Decision Making with Distributional Monte Carlo Tree Search","date":"2021-02-01","arxiv_id":"2102.00966","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-bicycle-dispatching-of-dockless","title":"Dynamic Bicycle Dispatching of Dockless Public Bicycle-sharing Systems using Multi-objective Reinforcement Learning","date":"2021-01-19","arxiv_id":"2101.07437","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximating-pareto-frontier-through","title":"Approximating Pareto Frontier through Bayesian-optimization-directed Robust Multi-objective Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-multi-objective-reinforcement","title":"Provable Multi-Objective Reinforcement Learning with Generative Models","date":"2020-11-19","arxiv_id":"2011.10134","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-based","title":"Multi-objective Reinforcement Learning based approach for User-Centric Power Optimization in Smart Home Environments","date":"2020-09-29","arxiv_id":"2009.13854","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributional-view-on-multi-objective-1","title":"A distributional view on multi objective policy optimization","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-logical-specifications-of-objectives-in","title":"Using Logical Specifications of Objectives in Multi-Objective Reinforcement Learning","date":"2019-10-03","arxiv_id":"1910.01723","repositories_listed":0,"syntology":null},{"url":null,"slug":"relationship-explainable-multi-objective","title":"Relationship Explainable Multi-objective Reinforcement Learning with Semantic Explainability Generation","date":"2019-09-26","arxiv_id":"1909.12268","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-single-objective-tasks-by-preference","title":"Solving single-objective tasks by preference multi-objective reinforcement learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-multi-objective","title":"Meta-Learning for Multi-objective Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03376","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-multi-objective-reinforcement","title":"Interpretable Multi-Objective Reinforcement Learning through Policy Orchestration","date":"2018-09-21","arxiv_id":"1809.08343","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-objective-deep-reinforcement-learning","title":"A Multi-Objective Deep Reinforcement Learning Framework","date":"2018-03-08","arxiv_id":"1803.02965","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-balancing-for-statistical-spoken","title":"Reward-Balancing for Statistical Spoken Dialogue Systems using Multi-objective Reinforcement Learning","date":"2017-07-19","arxiv_id":"1707.06299","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-negotiable-reinforcement-learning","title":"Toward negotiable reinforcement learning: shifting priorities in Pareto optimal sequential decision-making","date":"2017-01-05","arxiv_id":"1701.01302","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-with","title":"Multi-objective Reinforcement Learning with Continuous Pareto Frontier Approximation Supplementary Material","date":"2014-06-13","arxiv_id":"1406.3497","repositories_listed":0,"syntology":null}],"record_sha256":"aaacd62c47837d92ea3f21735b127397fbdd4a56db707d21b1ce0d159c042c59","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}