{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/87","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":87,"pages_in_order":135,"rows_per_page":100,"rows":[8601,8700],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/86","next":"/task/reinforcement-learning-2/papers/88","papers":[{"url":null,"slug":"offline-reinforcement-learning-fundamental","title":"Offline Reinforcement Learning: Fundamental Barriers for Value Function Approximation","date":"2021-11-21","arxiv_id":"2111.10919","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-general-ltl","title":"Reinforcement Learning with General LTL Objectives is Intractable","date":"2021-11-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"renewable-energy-integration-and-microgrid","title":"Renewable energy integration and microgrid energy trading using multi-agent deep reinforcement learning","date":"2021-11-21","arxiv_id":"2111.10898","repositories_listed":0,"syntology":null},{"url":null,"slug":"vulcan-solving-the-steiner-tree-problem-with","title":"Vulcan: Solving the Steiner Tree Problem with Graph Neural Networks and Deep Reinforcement Learning","date":"2021-11-21","arxiv_id":"2111.10810","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-biomedical-recommendations-via","title":"Explainable Biomedical Recommendations via Reinforcement Learning Reasoning on Knowledge Graphs","date":"2021-11-20","arxiv_id":"2111.10625","repositories_listed":0,"syntology":null},{"url":null,"slug":"rdf-to-text-generation-with-reinforcement","title":"Triples-to-Text Generation with Reinforcement Learning Based Graph-augmented Neural Networks","date":"2021-11-20","arxiv_id":"2111.10545","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-reinforcement-learning-model","title":"An Improved Reinforcement Learning Model Based on Sentiment Analysis","date":"2021-11-19","arxiv_id":"2111.15354","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-quasi-stationary-distributions-of","title":"Learn Quasi-stationary Distributions of Finite State Markov Chain","date":"2021-11-19","arxiv_id":"2111.11213","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-adaptive","title":"Reinforcement Learning with Adaptive Curriculum Dynamics Randomization for Fault-Tolerant Robot Control","date":"2021-11-19","arxiv_id":"2111.10005","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-low-rank-q-matrix","title":"Uncertainty-aware Low-Rank Q-Matrix Estimation for Deep Reinforcement Learning","date":"2021-11-19","arxiv_id":"2111.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-generalisation-in-deep","title":"A Survey of Zero-shot Generalisation in Deep Reinforcement Learning","date":"2021-11-18","arxiv_id":"2111.09794","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-reinforcement-learning-with-temporal","title":"Lifelong Reinforcement Learning with Temporal Logic Formulas and Reward Machines","date":"2021-11-18","arxiv_id":"2111.09475","repositories_listed":0,"syntology":null},{"url":null,"slug":"seihai-a-sample-efficient-hierarchical-ai-for","title":"SEIHAI: A Sample-efficient Hierarchical AI for the MineRL Competition","date":"2021-11-17","arxiv_id":"2111.08857","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-learning-tuning-for-post-silicon","title":"Self-Learning Tuning for Post-Silicon Validation","date":"2021-11-17","arxiv_id":"2111.08995","repositories_listed":0,"syntology":null},{"url":null,"slug":"clara-a-constrained-reinforcement-learning","title":"CLARA: A Constrained Reinforcement Learning Based Resource Allocation Framework for Network Slicing","date":"2021-11-16","arxiv_id":"2111.08397","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressive-features-in-offline-reinforcement","title":"Compressive Features in Offline Reinforcement Learning for Recommender Systems","date":"2021-11-16","arxiv_id":"2111.08817","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-entity","title":"Deep Reinforcement Learning for Entity Alignment","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mad-for-robust-reinforcement-learning-in","title":"MAD for Robust Reinforcement Learning in Machine Translation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"post-processing-networks-a-method-for","title":"Post-processing Networks: A Method for Optimizing Pipeline Task-oriented Dialogue Systems using Reinforcement Learning","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-the-robustness-of-trained-metrics-for","title":"Probing the Robustness of Trained Metrics for Conversational Dialogue Systems","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-feedback-from","title":"Reinforcement Learning with Feedback from Multiple Humans with Diverse Skills","date":"2021-11-16","arxiv_id":"2111.08596","repositories_listed":0,"syntology":null},{"url":null,"slug":"route-optimization-via-environment-aware-deep","title":"Route Optimization via Environment-Aware Deep Network and Reinforcement Learning","date":"2021-11-16","arxiv_id":"2111.09124","repositories_listed":0,"syntology":null},{"url":null,"slug":"delayed-feedback-in-episodic-reinforcement","title":"Optimism and Delays in Episodic Reinforcement Learning","date":"2021-11-15","arxiv_id":"2111.07615","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-action-impact-regularity-and","title":"Exploiting Action Impact Regularity and Exogenous State Variables for Offline Reinforcement Learning","date":"2021-11-15","arxiv_id":"2111.08066","repositories_listed":0,"syntology":null},{"url":null,"slug":"modellight-model-based-meta-reinforcement","title":"ModelLight: Model-Based Meta-Reinforcement Learning for Traffic Signal Control","date":"2021-11-15","arxiv_id":"2111.08067","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-partially-observable-history-process","title":"The Partially Observable History Process","date":"2021-11-15","arxiv_id":"2111.08102","repositories_listed":0,"syntology":null},{"url":null,"slug":"versatile-inverse-reinforcement-learning-via","title":"Versatile Inverse Reinforcement Learning via Cumulative Rewards","date":"2021-11-15","arxiv_id":"2111.07667","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualenv-visual-gym-environments-with","title":"VisualEnv: visual Gym environments with Blender","date":"2021-11-15","arxiv_id":"2111.08096","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-will-belief-as-a-consequence-of-model","title":"Free Will Belief as a consequence of Model-based Reinforcement Learning","date":"2021-11-14","arxiv_id":"2111.08435","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-distributed-formation-and-obstacle","title":"Relative Distributed Formation and Obstacle Avoidance with Multi-agent Reinforcement Learning","date":"2021-11-14","arxiv_id":"2111.07334","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-shallow","title":"Deep Reinforcement Learning with Shallow Controllers: An Experimental Application to PID Tuning","date":"2021-11-13","arxiv_id":"2111.07171","repositories_listed":0,"syntology":null},{"url":null,"slug":"obstacle-avoidance-for-uas-in-continuous","title":"Obstacle Avoidance for UAS in Continuous Action Space Using Deep Reinforcement Learning","date":"2021-11-13","arxiv_id":"2111.07037","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deep-reinforcement-learning-for-2","title":"Robust Deep Reinforcement Learning for Extractive Legal Summarization","date":"2021-11-13","arxiv_id":"2111.07158","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-multi-agent-reinforcement-learning","title":"Causal Multi-Agent Reinforcement Learning: Review and Open Problems","date":"2021-11-12","arxiv_id":"2111.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivergym-democratising-reinforcement","title":"DriverGym: Democratising Reinforcement Learning for Autonomous Driving","date":"2021-11-12","arxiv_id":"2111.06889","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlops-development-life-cycle-of-reinforcement","title":"RLOps: Development Life-cycle of Reinforcement Learning Aided Open RAN","date":"2021-11-12","arxiv_id":"2111.06978","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-surprise-minimizing-reinforcement","title":"Adapting Surprise Minimizing Reinforcement Learning Techniques for Transactive Control","date":"2021-11-11","arxiv_id":"2111.06025","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-spaces","title":"Agent Spaces","date":"2021-11-11","arxiv_id":"2111.06005","repositories_listed":0,"syntology":null},{"url":null,"slug":"cubetr-learning-to-solve-the-rubiks-cube","title":"CubeTR: Learning to Solve The Rubiks Cube Using Transformers","date":"2021-11-11","arxiv_id":"2111.06036","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-5","title":"Model-Based Reinforcement Learning via Stochastic Hybrid Models","date":"2021-11-11","arxiv_id":"2111.06211","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-8","title":"Multi-agent Reinforcement Learning for Cooperative Lane Changing of Connected and Autonomous Vehicles in Mixed Traffic","date":"2021-11-11","arxiv_id":"2111.06318","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-knowledge-graph-embedding-via","title":"Towards Robust Knowledge Graph Embedding via Multi-task Reinforcement Learning","date":"2021-11-11","arxiv_id":"2111.06103","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-before-you-leap-safe-model-based","title":"Look Before You Leap: Safe Model-Based Reinforcement Learning with Human Intervention","date":"2021-11-10","arxiv_id":"2111.05819","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatially-and-seamlessly-hierarchical-1","title":"Spatially and Seamlessly Hierarchical Reinforcement Learning for State Space and Policy space in Autonomous Driving","date":"2021-11-10","arxiv_id":"2111.05479","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealing-with-the-unknown-pessimistic-offline","title":"Dealing with the Unknown: Pessimistic Offline Reinforcement Learning","date":"2021-11-09","arxiv_id":"2111.05440","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-model-based-reinforcement","title":"Risk Sensitive Model-Based Reinforcement Learning using Uncertainty Guided Planning","date":"2021-11-09","arxiv_id":"2111.04972","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-reinforcement-learning-from-crowds","title":"Batch Reinforcement Learning from Crowds","date":"2021-11-08","arxiv_id":"2111.04279","repositories_listed":0,"syntology":null},{"url":null,"slug":"dueling-rl-reinforcement-learning-with","title":"Dueling RL: Reinforcement Learning with Trajectory Preferences","date":"2021-11-08","arxiv_id":"2111.04850","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-inverse-reinforcement-learning","title":"Interactive Inverse Reinforcement Learning for Cooperative Games","date":"2021-11-08","arxiv_id":"2111.04698","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-assessing-the-safety-of-reinforcement","title":"On Assessing The Safety of Reinforcement Learning algorithms Using Formal Methods","date":"2021-11-08","arxiv_id":"2111.04865","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-deep-reinforcement-learning-for-1","title":"Explainable Deep Reinforcement Learning for Portfolio Management: An Empirical Approach","date":"2021-11-07","arxiv_id":"2111.03995","repositories_listed":0,"syntology":null},{"url":null,"slug":"finrl-deep-reinforcement-learning-framework","title":"FinRL: Deep Reinforcement Learning Framework to Automate Trading in Quantitative Finance","date":"2021-11-07","arxiv_id":"2111.09395","repositories_listed":0,"syntology":null},{"url":null,"slug":"finrl-podracer-high-performance-and-scalable","title":"FinRL-Podracer: High Performance and Scalable Deep Reinforcement Learning for Quantitative Finance","date":"2021-11-07","arxiv_id":"2111.05188","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-the-model-predictive-control-1","title":"Optimization of the Model Predictive Control Meta-Parameters Through Reinforcement Learning","date":"2021-11-07","arxiv_id":"2111.04146","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-10","title":"A Deep Reinforcement Learning Approach for Composing Moving IoT Services","date":"2021-11-06","arxiv_id":"2111.03967","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-collective-behavior-in-newborn","title":"Development of collective behavior in newborn artificial agents","date":"2021-11-06","arxiv_id":"2111.03796","repositories_listed":0,"syntology":null},{"url":null,"slug":"exponential-bellman-equation-and-improved","title":"Exponential Bellman Equation and Improved Regret Bounds for Risk-Sensitive Reinforcement Learning","date":"2021-11-06","arxiv_id":"2111.03947","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rna-secondary-structure-design","title":"Improving RNA Secondary Structure Design using Deep Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.04504","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-cooperate-with-unseen-agent-via","title":"Learning to Cooperate with Unseen Agent via Meta-Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.03431","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacking-deep-reinforcement-learning-based","title":"Attacking Deep Reinforcement Learning-Based Traffic Signal Control Systems with Colluding Vehicles","date":"2021-11-04","arxiv_id":"2111.02845","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-a-fly-mimicking-flyer-in-complex","title":"Control of a fly-mimicking flyer in complex flow using deep reinforcement learning","date":"2021-11-04","arxiv_id":"2111.03454","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-in-dexterous-manipulation-via","title":"Generalization in Dexterous Manipulation via Geometry-Aware Multi-Task Learning","date":"2021-11-04","arxiv_id":"2111.03062","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagine-networks","title":"Imagine Networks","date":"2021-11-04","arxiv_id":"2111.03048","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-risk-sensitive-reinforcement","title":"Model-Free Risk-Sensitive Reinforcement Learning","date":"2021-11-04","arxiv_id":"2111.02907","repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-feature-neural-episodic-control","title":"Successor Feature Neural Episodic Control","date":"2021-11-04","arxiv_id":"2111.03110","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-learning-to-speak-and-hear-through-1","title":"Towards Learning to Speak and Hear Through Multi-Agent Communication over a Continuous Acoustic Channel","date":"2021-11-04","arxiv_id":"2111.02827","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-function-spaces-skill-centric-state-1","title":"Value Function Spaces: Skill-Centric State Abstractions for Long-Horizon Reasoning","date":"2021-11-04","arxiv_id":"2111.03189","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphad3m-machine-learning-pipeline-synthesis","title":"AlphaD3M: Machine Learning Pipeline Synthesis","date":"2021-11-03","arxiv_id":"2111.02508","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-attack-mitigation-for-industrial","title":"Autonomous Attack Mitigation for Industrial Control Systems","date":"2021-11-03","arxiv_id":"2111.02445","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-episodic-memory-induces-dynamic","title":"Model-Based Episodic Memory Induces Dynamic Hybrid Controls","date":"2021-11-03","arxiv_id":"2111.02104","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-service-provisioning-in-nfv-enabled","title":"Online Service Provisioning in NFV-enabled Networks Using Deep Reinforcement Learning","date":"2021-11-03","arxiv_id":"2111.02209","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-robot-do-i-need-fast-co-adaptation-of","title":"What Robot do I Need? Fast Co-Adaptation of Morphology and Control using Graph Neural Networks","date":"2021-11-03","arxiv_id":"2111.02371","repositories_listed":0,"syntology":null},{"url":"/paper/koopman-q-learning-offline-reinforcement-1","slug":"koopman-q-learning-offline-reinforcement-1","title":"Koopman Q-learning: Offline Reinforcement Learning via Symmetries of Dynamics","date":"2021-11-02","arxiv_id":"2111.01365","repositories_listed":0,"syntology":null},{"url":null,"slug":"onslicing-online-end-to-end-network-slicing","title":"OnSlicing: Online End-to-End Network Slicing with Reinforcement Learning","date":"2021-11-02","arxiv_id":"2111.01616","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-dynamic-bus-control-a-distributional","title":"Robust Dynamic Bus Control: A Distributional Multi-agent Reinforcement Learning Approach","date":"2021-11-02","arxiv_id":"2111.01946","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-collaborative-multi-agent-reinforcement-1","title":"A Collaborative Multi-agent Reinforcement Learning Framework for Dialog Action Decomposition","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generative-framework-for-simultaneous","title":"A Generative Framework for Simultaneous Machine Translation","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-cooperative-reinforcement","title":"Decentralized Cooperative Reinforcement Learning with Hierarchical Information Structure","date":"2021-11-01","arxiv_id":"2111.00781","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-independent-reinforcement","title":"Investigation of Independent Reinforcement Learning Algorithms in Multi-Agent Environments","date":"2021-11-01","arxiv_id":"2111.01100","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-task-sampling-policy-for-multitask","title":"Learning Task Sampling Policy for Multitask Learning","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-operate-an-electric-vehicle","title":"Learning to Operate an Electric Vehicle Charging Station Considering Vehicle-grid Integration","date":"2021-11-01","arxiv_id":"2111.01294","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-aided-crop-yield","title":"Machine Learning aided Crop Yield Optimization","date":"2021-11-01","arxiv_id":"2111.00963","repositories_listed":0,"syntology":null},{"url":null,"slug":"settling-the-horizon-dependence-of-sample","title":"Settling the Horizon-Dependence of Sample Complexity in Reinforcement Learning","date":"2021-11-01","arxiv_id":"2111.00633","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-multi-agent-reinforcement-3","title":"Decentralized Multi-Agent Reinforcement Learning: An Off-Policy Method","date":"2021-10-31","arxiv_id":"2111.00438","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-decentralized-reinforcement-learning","title":"A Decentralized Reinforcement Learning Framework for Efficient Passage of Emergency Vehicles","date":"2021-10-30","arxiv_id":"2111.00278","repositories_listed":0,"syntology":null},{"url":null,"slug":"adjacency-constraint-for-efficient","title":"Adjacency constraint for efficient hierarchical reinforcement learning","date":"2021-10-30","arxiv_id":"2111.00213","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-and-optimality-of-policy-gradient","title":"Convergence and Optimality of Policy Gradient Methods in Weakly Smooth Settings","date":"2021-10-30","arxiv_id":"2111.00185","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-discretization-in-online","title":"Adaptive Discretization in Online Reinforcement Learning","date":"2021-10-29","arxiv_id":"2110.15843","repositories_listed":0,"syntology":null},{"url":null,"slug":"brick-by-brick-combinatorial-construction","title":"Brick-by-Brick: Combinatorial Construction with Deep Reinforcement Learning","date":"2021-10-29","arxiv_id":"2110.15481","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-communicate-with-reinforcement","title":"Learning to Communicate with Reinforcement Learning for an Adaptive Traffic Control System","date":"2021-10-29","arxiv_id":"2110.15779","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-cooperative-competitive-communication","title":"Mixed Cooperative-Competitive Communication Using Multi-Agent Reinforcement Learning","date":"2021-10-29","arxiv_id":"2110.15762","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-robotic-reinforcement-learning","title":"Accelerating Robotic Reinforcement Learning via Parameterized Action Primitives","date":"2021-10-28","arxiv_id":"2110.15360","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-sequential-optimal-experimental","title":"Bayesian Sequential Optimal Experimental Design for Nonlinear Models Using Policy Gradient Reinforcement Learning","date":"2021-10-28","arxiv_id":"2110.15335","repositories_listed":0,"syntology":null},{"url":null,"slug":"choosing-the-best-of-both-worlds-diverse-and","title":"Choosing the Best of Both Worlds: Diverse and Novel Recommendations through Multi-Objective Reinforcement Learning","date":"2021-10-28","arxiv_id":"2110.15097","repositories_listed":0,"syntology":null},{"url":null,"slug":"d2rlir-an-improved-and-diversified-ranking","title":"D2RLIR : an improved and diversified ranking function in interactive recommendation systems based on deep reinforcement learning","date":"2021-10-28","arxiv_id":"2110.15089","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-clinician-s-goals-by-what-if","title":"Extracting Expert's Goals by What-if Interpretable Modeling","date":"2021-10-28","arxiv_id":"2110.15165","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-adaptive-dynamic-programming-for","title":"Data Informed Residual Reinforcement Learning for High-Dimensional Robotic Tracking Control","date":"2021-10-28","arxiv_id":"2110.15237","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-law-of-iterated-logarithm-for-multi-agent","title":"A Law of Iterated Logarithm for Multi-Agent Reinforcement Learning","date":"2021-10-27","arxiv_id":"2110.15092","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-subgame-perfect-equilibrium-reinforcement","title":"A Subgame Perfect Equilibrium Reinforcement Learning Approach to Time-inconsistent Problems","date":"2021-10-27","arxiv_id":"2110.14295","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-heuristics-constraint-optimization","title":"Comparing Heuristics, Constraint Optimization, and Reinforcement Learning for an Industrial 2D Packing Problem","date":"2021-10-27","arxiv_id":"2110.14535","repositories_listed":0,"syntology":null}],"record_sha256":"aadcdc159ca2e5604196739c21ca9f6daa951cd1c562f142a2ffb3f882da4ed6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}