{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/94","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":94,"pages_in_order":152,"rows_per_page":100,"rows":[9301,9400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/93","next":"/task/reinforcement-learning-1/papers/95","papers":[{"url":null,"slug":"wish-you-were-here-hindsight-goal-selection-1","title":"Wish you were here: Hindsight Goal Selection for long-horizon dexterous manipulation","date":"2021-12-01","arxiv_id":"2112.00597","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-design-for-a-noma-assisted","title":"Energy-Efficient Design for a NOMA assisted STAR-RIS Network with Deep Reinforcement Learning","date":"2021-11-30","arxiv_id":"2111.15464","repositories_listed":0,"syntology":null},{"url":null,"slug":"mamrl-exploiting-multi-agent-meta","title":"MAMRL: Exploiting Multi-agent Meta Reinforcement Learning in WAN Traffic Engineering","date":"2021-11-30","arxiv_id":"2111.15087","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-m-synthesis-via-adversarial","title":"Model-Free $μ$ Synthesis via Adversarial Reinforcement Learning","date":"2021-11-30","arxiv_id":"2111.15537","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-based-implementation-of-colregs-for","title":"Risk-based implementation of COLREGs for autonomous surface vehicles using deep reinforcement learning","date":"2021-11-30","arxiv_id":"2112.00115","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-power-of-communication-in-a-distributed","title":"The Power of Communication in a Distributed Multi-Agent System","date":"2021-11-30","arxiv_id":"2111.15611","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepcq-robust-and-scalable-routing-with-multi","title":"DeepCQ+: Robust and Scalable Routing with Multi-Agent Deep Reinforcement Learning for Highly Dynamic Networks","date":"2021-11-29","arxiv_id":"2111.15013","repositories_listed":0,"syntology":null},{"url":null,"slug":"final-adaptation-reinforcement-learning-for-n","title":"Final Adaptation Reinforcement Learning for N-Player Games","date":"2021-11-29","arxiv_id":"2111.14375","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-can-creativity-occur-in-multi-agent","title":"How Can Creativity Occur in Multi-Agent Systems?","date":"2021-11-29","arxiv_id":"2111.14310","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-zero-shot-generalization-in-offline-1","title":"Improving Zero-shot Generalization in Offline Reinforcement Learning using Generalized Similarity Functions","date":"2021-11-29","arxiv_id":"2111.14629","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-uav-conflict-resolution-with-graph","title":"Multi-UAV Conflict Resolution with Graph Convolutional Reinforcement Learning","date":"2021-11-29","arxiv_id":"2111.14598","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-model-selection-for-offline-deep-1","title":"Pessimistic Model Selection for Offline Deep Reinforcement Learning","date":"2021-11-29","arxiv_id":"2111.14346","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-algorithm-for-traffic","title":"Reinforcement Learning Algorithm for Traffic Steering in Heterogeneous Network","date":"2021-11-29","arxiv_id":"2111.15029","repositories_listed":0,"syntology":null},{"url":null,"slug":"count-based-temperature-scheduling-for","title":"Count-Based Temperature Scheduling for Maximum Entropy Reinforcement Learning","date":"2021-11-28","arxiv_id":"2111.14204","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-data-quality-for-dataset-selection","title":"Measuring Data Quality for Dataset Selection in Offline Reinforcement Learning","date":"2021-11-26","arxiv_id":"2111.13461","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-explanation-learning","title":"Reinforcement Explanation Learning","date":"2021-11-26","arxiv_id":"2111.13406","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-machine-learning-1","title":"A Comparative Analysis of Machine Learning Techniques for IoT Intrusion Detection","date":"2021-11-25","arxiv_id":"2111.13149","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepwive-deep-learning-aided-wireless-video","title":"DeepWiVe: Deep-Learning-Aided Wireless Video Transmission","date":"2021-11-25","arxiv_id":"2111.13034","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-policy-gradient-with-variance","title":"Distributed Policy Gradient with Variance Reduction in Multi-Agent Reinforcement Learning","date":"2021-11-25","arxiv_id":"2111.12961","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-challenges-for-reinforcement","title":"Real-world challenges for multi-agent reinforcement learning in grid-interactive buildings","date":"2021-11-25","arxiv_id":"2112.06127","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-modularity-optimization-using","title":"Towards Modularity Optimization Using Reinforcement Learning to Community Detection in Dynamic Social Networks","date":"2021-11-25","arxiv_id":"2111.15623","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comment-on-stabilizing-reinforcement","title":"A note on stabilizing reinforcement learning","date":"2021-11-24","arxiv_id":"2111.12316","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-for-deep-reinforcement-learning-in","title":"A Review for Deep Reinforcement Learning in Atari: Benchmarks, Challenges, and Solutions","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-multi-agent-reinforcement","title":"Application of Multi-Agent Reinforcement Learning for Battery Management in Renewable Mini-Grids","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/gdi-rethinking-what-makes-reinforcement-1","slug":"gdi-rethinking-what-makes-reinforcement-1","title":"GDI: Rethinking What Makes Reinforcement Learning Different from Supervised Learning","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-ai-play-football-an-analysis-of-rl","title":"How does AI play football? An analysis of RL and real-world football strategies","date":"2021-11-24","arxiv_id":"2111.12340","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-schedule-heuristics-for-the","title":"Learning to Schedule Heuristics for the Simultaneous Stochastic Optimization of Mining Complexes","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-path-exploration","title":"Reinforcement Learning based Path Exploration for Sequential Explainable Recommendation","date":"2021-11-24","arxiv_id":"2111.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-general-ltl","title":"On the (In)Tractability of Reinforcement Learning for LTL Objectives","date":"2021-11-24","arxiv_id":"2111.12679","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversible-action-design-for-combinatorial-1","title":"Reversible Action Design for Combinatorial Optimization with ReinforcementLearning","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"satnet-a-benchmark-for-satellite-scheduling","title":"SatNet: A Benchmark for Satellite Scheduling Optimization","date":"2021-11-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fixed-points-in-cyber-space-rethinking","title":"Fixed Points in Cyber Space: Rethinking Optimal Evasion Attacks in the Age of AI-NIDS","date":"2021-11-23","arxiv_id":"2111.12197","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-gpu-compiler-heuristics-using","title":"Generating GPU Compiler Heuristics using Reinforcement Learning","date":"2021-11-23","arxiv_id":"2111.12055","repositories_listed":0,"syntology":null},{"url":null,"slug":"independent-learning-in-stochastic-games","title":"Independent Learning in Stochastic Games","date":"2021-11-23","arxiv_id":"2111.11743","repositories_listed":0,"syntology":null},{"url":null,"slug":"inducing-functions-through-reinforcement","title":"Inducing Functions through Reinforcement Learning without Task Specification","date":"2021-11-23","arxiv_id":"2111.11647","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-volt-var-control-a","title":"Reinforcement Learning for Volt-Var Control: A Novel Two-stage Progressive Training Strategy","date":"2021-11-23","arxiv_id":"2111.11987","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-collaborative-deep","title":"Semantic-Aware Collaborative Deep Reinforcement Learning Over Wireless Cellular Networks","date":"2021-11-23","arxiv_id":"2111.12064","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbol-based-over-the-air-digital","title":"Symbol-Based Over-the-Air Digital Predistortion Using Reinforcement Learning","date":"2021-11-23","arxiv_id":"2111.11923","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-free-lunch-from-the-noise-provable-and-1","title":"A Free Lunch from the Noise: Provable and Practical Exploration for Representation Learning","date":"2021-11-22","arxiv_id":"2111.11485","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-application-of-reinforcement-learning-to","title":"An application of reinforcement learning to residential energy storage under real-time pricing","date":"2021-11-22","arxiv_id":"2111.11367","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-bayesian-inverse-reinforcement","title":"Efficient Bayesian Inverse Reinforcement Learning via Conditional Kernel Density Estimation","date":"2021-11-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-bayesian-deep-reinforcement","title":"Multi-agent Bayesian Deep Reinforcement Learning for Microgrid Energy Management under Communication Failures","date":"2021-11-22","arxiv_id":"2111.11868","repositories_listed":0,"syntology":null},{"url":null,"slug":"umbrella-uncertainty-aware-model-based","title":"UMBRELLA: Uncertainty-Aware Model-Based Offline Reinforcement Learning Leveraging Planning","date":"2021-11-22","arxiv_id":"2111.11097","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-neuro-symbolic-approach-for-text","title":"A Hybrid Neuro-Symbolic approach for Text-Based Games using Inductive Logic Programming","date":"2021-11-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-fundamental","title":"Offline Reinforcement Learning: Fundamental Barriers for Value Function Approximation","date":"2021-11-21","arxiv_id":"2111.10919","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-general-ltl","title":"Reinforcement Learning with General LTL Objectives is Intractable","date":"2021-11-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"renewable-energy-integration-and-microgrid","title":"Renewable energy integration and microgrid energy trading using multi-agent deep reinforcement learning","date":"2021-11-21","arxiv_id":"2111.10898","repositories_listed":0,"syntology":null},{"url":null,"slug":"vulcan-solving-the-steiner-tree-problem-with","title":"Vulcan: Solving the Steiner Tree Problem with Graph Neural Networks and Deep Reinforcement Learning","date":"2021-11-21","arxiv_id":"2111.10810","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-biomedical-recommendations-via","title":"Explainable Biomedical Recommendations via Reinforcement Learning Reasoning on Knowledge Graphs","date":"2021-11-20","arxiv_id":"2111.10625","repositories_listed":0,"syntology":null},{"url":null,"slug":"rdf-to-text-generation-with-reinforcement","title":"Triples-to-Text Generation with Reinforcement Learning Based Graph-augmented Neural Networks","date":"2021-11-20","arxiv_id":"2111.10545","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-safe-explainable-and-regulated","title":"Towards Safe, Explainable, and Regulated Autonomous Driving","date":"2021-11-20","arxiv_id":"2111.10518","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-reinforcement-learning-model","title":"An Improved Reinforcement Learning Model Based on Sentiment Analysis","date":"2021-11-19","arxiv_id":"2111.15354","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-quasi-stationary-distributions-of","title":"Learn Quasi-stationary Distributions of Finite State Markov Chain","date":"2021-11-19","arxiv_id":"2111.11213","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-for-mechanical-ventilation-1","title":"Machine Learning for Mechanical Ventilation Control (Extended Abstract)","date":"2021-11-19","arxiv_id":"2111.10434","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-adaptive","title":"Reinforcement Learning with Adaptive Curriculum Dynamics Randomization for Fault-Tolerant Robot Control","date":"2021-11-19","arxiv_id":"2111.10005","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-low-rank-q-matrix","title":"Uncertainty-aware Low-Rank Q-Matrix Estimation for Deep Reinforcement Learning","date":"2021-11-19","arxiv_id":"2111.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-generalisation-in-deep","title":"A Survey of Zero-shot Generalisation in Deep Reinforcement Learning","date":"2021-11-18","arxiv_id":"2111.09794","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-reinforcement-learning-with-temporal","title":"Lifelong Reinforcement Learning with Temporal Logic Formulas and Reward Machines","date":"2021-11-18","arxiv_id":"2111.09475","repositories_listed":0,"syntology":null},{"url":null,"slug":"seihai-a-sample-efficient-hierarchical-ai-for","title":"SEIHAI: A Sample-efficient Hierarchical AI for the MineRL Competition","date":"2021-11-17","arxiv_id":"2111.08857","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-learning-tuning-for-post-silicon","title":"Self-Learning Tuning for Post-Silicon Validation","date":"2021-11-17","arxiv_id":"2111.08995","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-document-coverage-reward-for-relaxed","title":"A Multi-Document Coverage Reward for RELAXed Multi-Document Summarization","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-policy-ranking","title":"Causal policy ranking","date":"2021-11-16","arxiv_id":"2111.08415","repositories_listed":0,"syntology":null},{"url":null,"slug":"clara-a-constrained-reinforcement-learning","title":"CLARA: A Constrained Reinforcement Learning Based Resource Allocation Framework for Network Slicing","date":"2021-11-16","arxiv_id":"2111.08397","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressive-features-in-offline-reinforcement","title":"Compressive Features in Offline Reinforcement Learning for Recommender Systems","date":"2021-11-16","arxiv_id":"2111.08817","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-language-modeling-for-goal","title":"Context-Aware Language Modeling for Goal-Oriented Dialogue Systems","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-entity","title":"Deep Reinforcement Learning for Entity Alignment","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"empathetic-persuasion-reinforcing-empathy-and","title":"Empathetic Persuasion: Reinforcing Empathy and Persuasiveness in Dialogue Systems","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-learning-from-demonstrations-by-1","title":"Improving Learning from Demonstrations by Learning from Experience","date":"2021-11-16","arxiv_id":"2111.08156","repositories_listed":0,"syntology":null},{"url":null,"slug":"mad-for-robust-reinforcement-learning-in","title":"MAD for Robust Reinforcement Learning in Machine Translation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"post-processing-networks-a-method-for","title":"Post-processing Networks: A Method for Optimizing Pipeline Task-oriented Dialogue Systems using Reinforcement Learning","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-the-robustness-of-trained-metrics-for","title":"Probing the Robustness of Trained Metrics for Conversational Dialogue Systems","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-feedback-from","title":"Reinforcement Learning with Feedback from Multiple Humans with Diverse Skills","date":"2021-11-16","arxiv_id":"2111.08596","repositories_listed":0,"syntology":null},{"url":null,"slug":"route-optimization-via-environment-aware-deep","title":"Route Optimization via Environment-Aware Deep Network and Reinforcement Learning","date":"2021-11-16","arxiv_id":"2111.09124","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-skill-chaining-for-long-horizon","title":"Adversarial Skill Chaining for Long-Horizon Robot Manipulation via Terminal State Regularization","date":"2021-11-15","arxiv_id":"2111.07999","repositories_listed":0,"syntology":null},{"url":null,"slug":"common-language-for-goal-oriented-semantic","title":"Common Language for Goal-Oriented Semantic Communications: A Curriculum Learning Framework","date":"2021-11-15","arxiv_id":"2111.08051","repositories_listed":0,"syntology":null},{"url":null,"slug":"delayed-feedback-in-episodic-reinforcement","title":"Optimism and Delays in Episodic Reinforcement Learning","date":"2021-11-15","arxiv_id":"2111.07615","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-action-impact-regularity-and","title":"Exploiting Action Impact Regularity and Exogenous State Variables for Offline Reinforcement Learning","date":"2021-11-15","arxiv_id":"2111.08066","repositories_listed":0,"syntology":null},{"url":null,"slug":"modellight-model-based-meta-reinforcement","title":"ModelLight: Model-Based Meta-Reinforcement Learning for Traffic Signal Control","date":"2021-11-15","arxiv_id":"2111.08067","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-partially-observable-history-process","title":"The Partially Observable History Process","date":"2021-11-15","arxiv_id":"2111.08102","repositories_listed":0,"syntology":null},{"url":null,"slug":"versatile-inverse-reinforcement-learning-via","title":"Versatile Inverse Reinforcement Learning via Cumulative Rewards","date":"2021-11-15","arxiv_id":"2111.07667","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualenv-visual-gym-environments-with","title":"VisualEnv: visual Gym environments with Blender","date":"2021-11-15","arxiv_id":"2111.08096","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-explore-exploit-or-escape-e-4-near","title":"Explicit Explore, Exploit, or Escape ($E^4$): near-optimal safety-constrained reinforcement learning in polynomial time","date":"2021-11-14","arxiv_id":"2111.07395","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-will-belief-as-a-consequence-of-model","title":"Free Will Belief as a consequence of Model-based Reinforcement Learning","date":"2021-11-14","arxiv_id":"2111.08435","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-distributed-formation-and-obstacle","title":"Relative Distributed Formation and Obstacle Avoidance with Multi-agent Reinforcement Learning","date":"2021-11-14","arxiv_id":"2111.07334","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-shallow","title":"Deep Reinforcement Learning with Shallow Controllers: An Experimental Application to PID Tuning","date":"2021-11-13","arxiv_id":"2111.07171","repositories_listed":0,"syntology":null},{"url":null,"slug":"obstacle-avoidance-for-uas-in-continuous","title":"Obstacle Avoidance for UAS in Continuous Action Space Using Deep Reinforcement Learning","date":"2021-11-13","arxiv_id":"2111.07037","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deep-reinforcement-learning-for-2","title":"Robust Deep Reinforcement Learning for Extractive Legal Summarization","date":"2021-11-13","arxiv_id":"2111.07158","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-to-look-a-unified-attention-model-for","title":"Where to Look: A Unified Attention Model for Visual Recognition with Reinforcement Learning","date":"2021-11-13","arxiv_id":"2111.07169","repositories_listed":0,"syntology":null},{"url":null,"slug":"awd3-dynamic-reduction-of-the-estimation-bias","title":"AWD3: Dynamic Reduction of the Estimation Bias","date":"2021-11-12","arxiv_id":"2111.06780","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-multi-agent-reinforcement-learning","title":"Causal Multi-Agent Reinforcement Learning: Review and Open Problems","date":"2021-11-12","arxiv_id":"2111.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivergym-democratising-reinforcement","title":"DriverGym: Democratising Reinforcement Learning for Autonomous Driving","date":"2021-11-12","arxiv_id":"2111.06889","repositories_listed":0,"syntology":null},{"url":null,"slug":"promoting-resilience-in-multi-agent","title":"Collaboration Promotes Group Resilience in Multi-Agent AI","date":"2021-11-12","arxiv_id":"2111.06614","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlops-development-life-cycle-of-reinforcement","title":"RLOps: Development Life-cycle of Reinforcement Learning Aided Open RAN","date":"2021-11-12","arxiv_id":"2111.06978","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-surprise-minimizing-reinforcement","title":"Adapting Surprise Minimizing Reinforcement Learning Techniques for Transactive Control","date":"2021-11-11","arxiv_id":"2111.06025","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-spaces","title":"Agent Spaces","date":"2021-11-11","arxiv_id":"2111.06005","repositories_listed":0,"syntology":null},{"url":null,"slug":"cubetr-learning-to-solve-the-rubiks-cube","title":"CubeTR: Learning to Solve The Rubiks Cube Using Transformers","date":"2021-11-11","arxiv_id":"2111.06036","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-5","title":"Model-Based Reinforcement Learning via Stochastic Hybrid Models","date":"2021-11-11","arxiv_id":"2111.06211","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-8","title":"Multi-agent Reinforcement Learning for Cooperative Lane Changing of Connected and Autonomous Vehicles in Mixed Traffic","date":"2021-11-11","arxiv_id":"2111.06318","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-knowledge-graph-embedding-via","title":"Towards Robust Knowledge Graph Embedding via Multi-task Reinforcement Learning","date":"2021-11-11","arxiv_id":"2111.06103","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-before-you-leap-safe-model-based","title":"Look Before You Leap: Safe Model-Based Reinforcement Learning with Human Intervention","date":"2021-11-10","arxiv_id":"2111.05819","repositories_listed":0,"syntology":null}],"record_sha256":"fa9ce3712aea903139837fa7a509eda9825d517104667a035a3f6f8e57ff0b3f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}