{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/107","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":107,"pages_in_order":152,"rows_per_page":100,"rows":[10601,10700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/106","next":"/task/reinforcement-learning-1/papers/108","papers":[{"url":null,"slug":"bayesian-distributional-policy-gradients","title":"Bayesian Distributional Policy Gradients","date":"2021-03-20","arxiv_id":"2103.11265","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-pessimism-with-optimism-for-robust","title":"Combining Pessimism with Optimism for Robust and Efficient Model-Based Deep Reinforcement Learning","date":"2021-03-18","arxiv_id":"2103.10369","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-aided-ran-slicing","title":"Deep Reinforcement Learning-Aided RAN Slicing Enforcement for B5G Latency Sensitive Services","date":"2021-03-18","arxiv_id":"2103.10277","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-synthesis-for-data-augmentation-in","title":"Image Synthesis for Data Augmentation in Medical CT using Deep Reinforcement Learning","date":"2021-03-18","arxiv_id":"2103.10493","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-reinforcement-learning-with","title":"Maximum Entropy Reinforcement Learning with Mixture Policies","date":"2021-03-18","arxiv_id":"2103.10176","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-reinforcement-learning-for","title":"Decentralized Reinforcement Learning for Multi-Target Search and Detection by a Team of Drones","date":"2021-03-17","arxiv_id":"2103.09520","repositories_listed":0,"syntology":null},{"url":null,"slug":"infinite-horizon-offline-reinforcement","title":"Infinite-Horizon Offline Reinforcement Learning with Linear Function Approximation: Curse of Dimensionality and Algorithm","date":"2021-03-17","arxiv_id":"2103.09847","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-policy-optimization-via-reps","title":"Near Optimal Policy Optimization via REPS","date":"2021-03-17","arxiv_id":"2103.09756","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-framework-1","title":"Hierarchical Reinforcement Learning Framework for Stochastic Spaceflight Campaign Design","date":"2021-03-16","arxiv_id":"2103.08981","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-shape-rewards-using-a-game-of","title":"Learning to Shape Rewards using a Game of Two Partners","date":"2021-03-16","arxiv_id":"2103.09159","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-curriculum-reinforcement-learning-for","title":"Goal-constrained Sparse Reinforcement Learning for End-to-End Driving","date":"2021-03-16","arxiv_id":"2103.09189","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-online-reinforcement-learning-1","title":"Accelerating Online Reinforcement Learning via Model-Based Meta-Learning","date":"2021-03-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-drone-racing-with-deep","title":"Autonomous Drone Racing with Deep Reinforcement Learning","date":"2021-03-15","arxiv_id":"2103.08624","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-symbolic-rules-for-interpretable","title":"Learning Symbolic Rules for Interpretable Deep Reinforcement Learning","date":"2021-03-15","arxiv_id":"2103.08228","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-human-kinetics-and-kinematics","title":"Modelling Human Kinetics and Kinematics during Walking using Reinforcement Learning","date":"2021-03-15","arxiv_id":"2103.08125","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-value-of-curriculum","title":"Investigating Value of Curriculum Reinforcement Learning in Autonomous Driving Under Diverse Road and Weather Conditions","date":"2021-03-14","arxiv_id":"2103.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulation-studies-on-deep-reinforcement","title":"Simulation Studies on Deep Reinforcement Learning for Building Control with Human Interaction","date":"2021-03-14","arxiv_id":"2103.07919","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-forex-and-stock-price-prediction","title":"A Survey of Forex and Stock Price Prediction Using Deep Learning","date":"2021-03-13","arxiv_id":"2103.09750","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-computer-approach-to-train-a-machine","title":"Hybrid computer approach to train a machine learning system","date":"2021-03-13","arxiv_id":"2103.07802","repositories_listed":0,"syntology":null},{"url":null,"slug":"metalearning-using-structure-rich-pipeline","title":"Metalearning Using Structure-rich Pipeline Representations for Better AutoML","date":"2021-03-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-controller-a-reinforcement-learning","title":"RL-Controller: a reinforcement learning framework for active structural control","date":"2021-03-13","arxiv_id":"2103.07616","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-text-generation-with-global","title":"Constrained Text Generation with Global Guidance -- Case Study on CommonGen","date":"2021-03-12","arxiv_id":"2103.07170","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-approach-to","title":"A Reinforcement Learning Based Approach to Play Calling in Football","date":"2021-03-11","arxiv_id":"2103.06939","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-vision-based-deep-reinforcement-learning","title":"A Vision Based Deep Reinforcement Learning Algorithm for UAV Obstacle Avoidance","date":"2021-03-11","arxiv_id":"2103.06403","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-user-interfaces-with-model-based","title":"Adapting User Interfaces with Model-based Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06807","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-in-consensus-based-multi","title":"Adversarial attacks in consensus-based multi-agent reinforcement learning","date":"2021-03-11","arxiv_id":"2103.06967","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-hidden-activations-of-deep","title":"Analyzing the Hidden Activations of Deep Policy Networks: Why Representation Matters","date":"2021-03-11","arxiv_id":"2103.06398","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-cop-adaptation-generation-in-context","title":"Auto-COP: Adaptation Generation in Context-Oriented Programming using Reinforcement Learning Options","date":"2021-03-11","arxiv_id":"2103.06757","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-federated-reinforcement-learning","title":"Multi-Task Federated Reinforcement Learning with Adversaries","date":"2021-03-11","arxiv_id":"2103.06473","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-finite-sample-analysis-of-offline","title":"Sample Complexity of Offline Reinforcement Learning with Deep ReLU Networks","date":"2021-03-11","arxiv_id":"2103.06671","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-high-speed-running-for-quadruped","title":"Robust High-speed Running for Quadruped Robots via Deep Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06484","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-reinforcement-learning-for-safe-ran","title":"Symbolic Reinforcement Learning for Safe RAN Control","date":"2021-03-11","arxiv_id":"2103.06602","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-stage-framework-and-reinforcement","title":"A Two-stage Framework and Reinforcement Learning-based Optimization Algorithms for Complex Scheduling Problems","date":"2021-03-10","arxiv_id":"2103.05847","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-perspective-on-1","title":"An Information-Theoretic Perspective on Credit Assignment in Reinforcement Learning","date":"2021-03-10","arxiv_id":"2103.06224","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-gradient-dqn-reinforcement-learning-a","title":"Full Gradient DQN Reinforcement Learning: A Provably Convergent Scheme","date":"2021-03-10","arxiv_id":"2103.05981","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-context-based-meta-reinforcement","title":"Improving Context-Based Meta-Reinforcement Learning with Self-Supervised Trajectory Contrastive Learning","date":"2021-03-10","arxiv_id":"2103.06386","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-of-autonomous","title":"WFA-IRL: Inverse Reinforcement Learning of Autonomous Behaviors Encoded as Weighted Finite Automata","date":"2021-03-10","arxiv_id":"2103.05895","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-rl-provably-solves-some","title":"Maximum Entropy RL (Provably) Solves Some Robust RL Problems","date":"2021-03-10","arxiv_id":"2103.06257","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-based-1","title":"Multi-Objective Reinforcement Learning based Multi-Microgrid System Optimisation Problem","date":"2021-03-10","arxiv_id":"2103.06380","repositories_listed":0,"syntology":null},{"url":null,"slug":"s4rl-surprisingly-simple-self-supervision-for","title":"S4RL: Surprisingly Simple Self-Supervision for Offline Reinforcement Learning","date":"2021-03-10","arxiv_id":"2103.06326","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-linear-system-identification-with","title":"Streaming Linear System Identification with Reverse Experience Replay","date":"2021-03-10","arxiv_id":"2103.05896","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-cognitive-models-to-train-warm-start","title":"Using Cognitive Models to Train Warm Start Reinforcement Learning Agents for Human-Computer Interactions","date":"2021-03-10","arxiv_id":"2103.06160","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-goal-generation-using-dynamical-1","title":"Automatic Goal Generation using Dynamical Distance Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-for-reinforcement-learning-in","title":"Challenges for Reinforcement Learning in Healthcare","date":"2021-03-09","arxiv_id":"2103.05612","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-impact-time-guidance-a-learning","title":"A Learning-Based Computational Impact Time Guidance","date":"2021-03-09","arxiv_id":"2103.05196","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-circle-formation-control-for","title":"Decentralized Circle Formation Control for Fish-like Robots in the Real-world via Reinforcement Learning","date":"2021-03-09","arxiv_id":"2103.05293","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-am-robot-neuromuscular-reinforcement","title":"I am Robot: Neuromuscular Reinforcement Learning to Actuate Human Limbs through Functional Electrical Stimulation","date":"2021-03-09","arxiv_id":"2103.05349","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-state-representations-via-temporal","title":"Learning State Representations via Temporal Cycle-Consistency Constraint in Model-Based Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-task-informed-abstractions","title":"Learning Task Informed Abstractions","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-a-class-of-multiple","title":"Learning to Explore a Class of Multiple Reward-Free Environments","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-infer-unseen-contexts-in-causal","title":"Learning to Infer Unseen Contexts in Causal Contextual Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"less-suboptimal-learning-and-control-in","title":"Less Suboptimal Learning and Control in Variational POMDPs","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"loco-adaptive-exploration-in-reinforcement","title":"LOCO: Adaptive exploration in reinforcement learning via local estimation of contraction coefficients","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"minimum-description-length-skills-for","title":"Minimum Description Length Skills for Accelerated Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"out-of-distribution-generalization-of","title":"Out-of-distribution generalization of internal models is correlated with reward","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pretraining-reward-free-representations-for","title":"Pretraining Reward-Free Representations for Data-Efficient Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"psiphi-learning-reinforcement-learning-with-1","title":"PsiPhi-Learning: Reinforcement Learning with Demonstrations using Successor Features and Inverse Temporal Difference Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"resolving-causal-confusion-in-reinforcement","title":"Resolving Causal Confusion in Reinforcement Learning via Robust Exploration","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"solipsistic-reinforcement-learning","title":"Solipsistic Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-quantum-policies-for","title":"Parametrized quantum policies for reinforcement learning","date":"2021-03-09","arxiv_id":"2103.05577","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-taxonomy-of-similarity-metrics-for-markov","title":"A Taxonomy of Similarity Metrics for Markov Decision Processes","date":"2021-03-08","arxiv_id":"2103.04706","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-for-1","title":"Adversarial Reinforcement Learning for Procedural Content Generation","date":"2021-03-08","arxiv_id":"2103.04847","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-energy-saving-snake-locomotion-gait-policy","title":"An Energy-Saving Snake Locomotion Gait Policy Obtained Using Deep Reinforcement Learning","date":"2021-03-08","arxiv_id":"2103.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-models-the","title":"A multi-agent reinforcement learning model of reputation and cooperation in human groups","date":"2021-03-08","arxiv_id":"2103.04982","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-reinforcement-learning-for-2","title":"Distributed Reinforcement Learning for Flexible and Efficient UAV Swarm Control","date":"2021-03-08","arxiv_id":"2103.04666","repositories_listed":0,"syntology":null},{"url":null,"slug":"increasing-energy-efficiency-of-massive-mimo","title":"Increasing Energy Efficiency of Massive-MIMO Network via Base Stations Switching using Reinforcement Learning and Radio Environment Maps","date":"2021-03-08","arxiv_id":"2103.11891","repositories_listed":0,"syntology":null},{"url":null,"slug":"instabilities-of-offline-rl-with-pre-trained","title":"Instabilities of Offline RL with Pre-Trained Neural Representation","date":"2021-03-08","arxiv_id":"2103.04947","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-cooperative-multi-agent","title":"Provably Efficient Cooperative Multi-Agent Reinforcement Learning with Function Approximation","date":"2021-03-08","arxiv_id":"2103.04972","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-ride-hailing-vehicle-repositioning","title":"Real-world Ride-hailing Vehicle Repositioning using Deep Reinforcement Learning","date":"2021-03-08","arxiv_id":"2103.04555","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-rlr-tree-a-reinforcement-learning-based-r","title":"A Reinforcement Learning Based R-Tree for Spatial Data Indexing in Dynamic Environments","date":"2021-03-08","arxiv_id":"2103.04541","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-based-mobile-robotics-obstacle","title":"Vision-Based Mobile Robotics Obstacle Avoidance With Deep Reinforcement Learning","date":"2021-03-08","arxiv_id":"2103.04727","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lower-bound-for-the-sample-complexity-of","title":"A Lower Bound for the Sample Complexity of Inverse Reinforcement Learning","date":"2021-03-07","arxiv_id":"2103.04446","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-human-rewards-by-inferring-their","title":"Learning Human Rewards by Inferring Their Latent Intelligence Levels in Multi-Agent Games: A Theory-of-Mind Approach with Application to Driving Data","date":"2021-03-07","arxiv_id":"2103.04289","repositories_listed":0,"syntology":null},{"url":null,"slug":"markov-cricket-using-forward-and-inverse","title":"Markov Cricket: Using Forward and Inverse Reinforcement Learning to Model, Predict And Optimize Batting Performance in One-Day International Cricket","date":"2021-03-07","arxiv_id":"2103.04349","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-reinforcement-learning-an-instrumental","title":"Asymptotic Theory for IV-Based Reinforcement Learning with Potential Endogeneity","date":"2021-03-06","arxiv_id":"2103.04021","repositories_listed":0,"syntology":null},{"url":null,"slug":"passing-through-narrow-gaps-with-deep","title":"Passing Through Narrow Gaps with Deep Reinforcement Learning","date":"2021-03-06","arxiv_id":"2103.03991","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-bit-by-bit","title":"Reinforcement Learning, Bit by Bit","date":"2021-03-06","arxiv_id":"2103.04047","repositories_listed":0,"syntology":null},{"url":null,"slug":"saint-acc-safety-aware-intelligent-adaptive","title":"SAINT-ACC: Safety-Aware Intelligent Adaptive Cruise Control for Autonomous Vehicles Using Deep Reinforcement Learning","date":"2021-03-06","arxiv_id":"2104.06506","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-explanation-using-attention-mechanism-1","title":"Visual Explanation using Attention Mechanism in Actor-Critic-based Deep Reinforcement Learning","date":"2021-03-06","arxiv_id":"2103.04067","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-memory-architecture-for-reinforcement","title":"A Dual-Memory Architecture for Reinforcement Learning on Neuromorphic Platforms","date":"2021-03-05","arxiv_id":"2103.04780","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-exploration-process-adjustment-for","title":"Automatic Exploration Process Adjustment for Safe Reinforcement Learning with Joint Chance Constraint Satisfaction","date":"2021-03-05","arxiv_id":"2103.03656","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-medical","title":"Deep reinforcement learning in medical imaging: A literature review","date":"2021-03-05","arxiv_id":"2103.05115","repositories_listed":0,"syntology":null},{"url":null,"slug":"routing-algorithms-as-tools-for-integrating","title":"Routing algorithms as tools for integrating social distancing with emergency evacuation","date":"2021-03-05","arxiv_id":"2103.03413","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-rl-based-adaptive-detection-strategy-to","title":"An RL-Based Adaptive Detection Strategy to Secure Cyber-Physical Systems","date":"2021-03-04","arxiv_id":"2103.02872","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-explicit","title":"Inverse Reinforcement Learning with Explicit Policy Estimates","date":"2021-03-04","arxiv_id":"2103.02863","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuromechanics-based-deep-reinforcement","title":"Neuromechanics-based Deep Reinforcement Learning of Neurostimulation Control in FES cycling","date":"2021-03-04","arxiv_id":"2103.03057","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-uav-trajectory-planning-using","title":"Efficient UAV Trajectory-Planning using Economic Reinforcement Learning","date":"2021-03-03","arxiv_id":"2103.02676","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-orientation","title":"Reinforcement Learning for Orientation Estimation Using Inertial Sensors with Performance Guarantee","date":"2021-03-03","arxiv_id":"2103.02357","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-external-1","title":"Reinforcement Learning with External Knowledge by using Logical Neural Networks","date":"2021-03-03","arxiv_id":"2103.02363","repositories_listed":0,"syntology":null},{"url":null,"slug":"shape-driven-coordinate-ordering-for-star","title":"Shape-driven Coordinate Ordering for Star Glyph Sets via Reinforcement Learning","date":"2021-03-03","arxiv_id":"2103.02380","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-model-learning","title":"Minimax Model Learning","date":"2021-03-02","arxiv_id":"2103.02084","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with","title":"Offline Reinforcement Learning with Pseudometric Learning","date":"2021-03-02","arxiv_id":"2103.01948","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-navigation-of-an-ultrasound-probe","title":"Autonomous Navigation of an Ultrasound Probe Towards Standard Scan Planes with Deep Reinforcement Learning","date":"2021-03-01","arxiv_id":"2103.00718","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-monopoly-gameplay-a-hybrid-model","title":"Decision Making in Monopoly using a Hybrid Deep Reinforcement Learning Approach","date":"2021-03-01","arxiv_id":"2103.00683","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-adaptive-mesh","title":"Reinforcement Learning for Adaptive Mesh Refinement","date":"2021-03-01","arxiv_id":"2103.01342","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-and-incentives-in-reinforcement","title":"Exploration and Incentives in Reinforcement Learning","date":"2021-02-28","arxiv_id":"2103.00360","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-policy-optimization","title":"Hamiltonian Policy Optimization","date":"2021-02-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-for-visual-navigation-by-imagining","title":"Learning for Visual Navigation by Imagining the Success","date":"2021-02-28","arxiv_id":"2103.00446","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-the-action-is-let-s-make-reinforcement","title":"Where the Action is: Let's make Reinforcement Learning for Stochastic Dynamic Vehicle Routing Problems work!","date":"2021-02-28","arxiv_id":"2103.00507","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-control-of-point-to-point-navigation","title":"Optimal control of point-to-point navigation in turbulent time-dependent flows using Reinforcement Learning","date":"2021-02-27","arxiv_id":"2103.00329","repositories_listed":0,"syntology":null}],"record_sha256":"3933ca3c8e66fa0e12b942d55fa3e12bebc55de40069238ea41620ad0a1e9b7a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}