{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/distributional-reinforcement-learning/papers/2","list_of":"/task/distributional-reinforcement-learning","task":"Distributional Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,137],"of":137,"counts":{"archive_papers_tagged":137,"with_a_code_link":43,"where_syntology_ran_a_sample":16,"not_listed_spam_title":0,"listed":137,"listed_where_code_ran":16,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":12,"every_run_a_failure_of_syntologys_instrument":4,"listed_with_a_run_with_no_instrument_failure":12,"listed_every_run_a_failure_of_syntologys_instrument":4,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/distributional-reinforcement-learning","prev":"/task/distributional-reinforcement-learning","next":null,"papers":[{"url":null,"slug":"conservative-distributional-reinforcement","title":"Conservative Distributional Reinforcement Learning with Safety Constraints","date":"2022-01-18","arxiv_id":"2201.07286","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-and-risk-management-via","title":"Robustness and risk management via distributional dynamic programming","date":"2021-12-28","arxiv_id":"2112.15430","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-distributional","title":"The Benefits of Being Categorical Distributional: Uncertainty-aware Regularized Exploration in Reinforcement Learning","date":"2021-10-07","arxiv_id":"2110.03155","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-perturbation-for-efficient","title":"Distributional Perturbation for Efficient Exploration in Distributional Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-5","title":"Distributional Reinforcement Learning with Monotonic Splines","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-robustness-of-distributional-1","title":"Exploring the Robustness of Distributional Reinforcement Learning against Noisy State Observations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-distributional-1","title":"Towards Understanding Distributional Reinforcement Learning: Regularization, Optimization, Acceleration and Sinkhorn Algorithm","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"minimizing-safety-interference-for-safe-and","title":"Minimizing Safety Interference for Safe and Comfortable Automated Driving with Distributional Reinforcement Learning","date":"2021-07-15","arxiv_id":"2107.07316","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmd-mix-value-function-factorisation-with","title":"MMD-MIX: Value Function Factorisation with Maximum Mean Discrepancy for Cooperative Multi-Agent Reinforcement Learning","date":"2021-06-22","arxiv_id":"2106.11652","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-risk-adaptation-in-distributional","title":"Automatic Risk Adaptation in Distributional Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06317","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-decreasing-quantile-function-network-with-1","title":"Non-decreasing Quantile Function Network with Efficient Exploration for Distributional Reinforcement Learning","date":"2021-05-14","arxiv_id":"2105.06696","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-distributional-policy-gradients","title":"Bayesian Distributional Policy Gradients","date":"2021-03-20","arxiv_id":"2103.11265","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-distributional-reinforcement-learning","title":"Safe Distributional Reinforcement Learning","date":"2021-02-26","arxiv_id":"2102.13446","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentinel-taming-uncertainty-with-ensemble","title":"SENTINEL: Taming Uncertainty with Ensemble-based Distributional Reinforcement Learning","date":"2021-02-22","arxiv_id":"2102.11075","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-inherent-uncertainty-risk","title":"Addressing Inherent Uncertainty: Risk-Sensitive Behavior Generation for Automated Driving using Distributional Reinforcement Learning","date":"2021-02-05","arxiv_id":"2102.03119","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-synthetic-characters-in","title":"Controlling Synthetic Characters in Simulations: A Case for Cognitive Architectures and Sigma","date":"2021-01-06","arxiv_id":"2101.02231","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributional-perspective-on-actor-critic","title":"A Distributional Perspective on Actor-Critic Framework","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-for-2","title":"Distributional Reinforcement Learning for Risk-Sensitive Policies","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-constrained-distributional","title":"Batch-Constrained Distributional Reinforcement Learning for Session-based Recommendation","date":"2020-12-16","arxiv_id":"2012.08984","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-local-temporal-difference-code-for","title":"A Local Temporal Difference Code for Distributional Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"non-crossing-quantile-regression-for","title":"Non-Crossing Quantile Regression for Distributional Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-for-3","title":"Distributional Reinforcement Learning for mmWave Communications with Intelligent Reflectors on a UAV","date":"2020-11-03","arxiv_id":"2011.01840","repositories_listed":0,"syntology":null},{"url":null,"slug":"demand-side-scheduling-based-on-deep-actor","title":"Demand-Side Scheduling Based on Multi-Agent Deep Actor-Critic Learning for Smart Grids","date":"2020-05-05","arxiv_id":"2005.01979","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-robustness-via-risk-averse","title":"Improving Robustness via Risk Averse Distributional Reinforcement Learning","date":"2020-05-01","arxiv_id":"2005.00585","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-2","title":"Distributional Reinforcement Learning with Ensembles","date":"2020-03-24","arxiv_id":"2003.10903","repositories_listed":0,"syntology":null},{"url":null,"slug":"millimeter-wave-communications-with-an","title":"Millimeter Wave Communications with an Intelligent Reflector: Performance Optimization and Distributional Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10572","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-based-distributional-policy-gradient","title":"Sample-based Distributional Policy Gradient","date":"2020-01-08","arxiv_id":"2001.02652","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastically-dominant-distributional","title":"Stochastically Dominant Distributional Reinforcement Learning","date":"2019-05-17","arxiv_id":"1905.07318","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-for","title":"Distributional Reinforcement Learning for Efficient Exploration","date":"2019-05-13","arxiv_id":"1905.06125","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-based-deep-distributional-reinforcement","title":"GAN-powered Deep Distributional Reinforcement Learning for Resource Management in Network Slicing","date":"2019-05-10","arxiv_id":"1905.03929","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistics-and-samples-in-distributional","title":"Statistics and Samples in Distributional Reinforcement Learning","date":"2019-02-21","arxiv_id":"1902.08102","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with","title":"Distributional reinforcement learning with linear function approximation","date":"2019-02-08","arxiv_id":"1902.03149","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-expected-and","title":"A Comparative Analysis of Expected and Distributional Reinforcement Learning","date":"2019-01-30","arxiv_id":"1901.11084","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonlinear-distributional-gradient-temporal","title":"Nonlinear Distributional Gradient Temporal-Difference Learning","date":"2018-05-20","arxiv_id":"1805.07732","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-by-distributional-reinforcement","title":"Exploration by Distributional Reinforcement Learning","date":"2018-05-04","arxiv_id":"1805.01907","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-categorical-distributional","title":"An Analysis of Categorical Distributional Reinforcement Learning","date":"2018-02-22","arxiv_id":"1802.08163","repositories_listed":0,"syntology":null},{"url":"/paper/the-reactor-a-fast-and-sample-efficient-actor","slug":"the-reactor-a-fast-and-sample-efficient-actor","title":"The Reactor: A fast and sample-efficient Actor-Critic agent for Reinforcement Learning","date":"2017-04-15","arxiv_id":"1704.04651","repositories_listed":0,"syntology":null}],"record_sha256":"2a7ede81e4d192f3b5b57b7d458474343a361308007e7091e54312230ee7d3ca","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}