{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/114","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":114,"pages_in_order":152,"rows_per_page":100,"rows":[11301,11400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/113","next":"/task/reinforcement-learning-1/papers/115","papers":[{"url":null,"slug":"hierarchical-reinforcement-learning-for-3","title":"Hierarchical reinforcement learning for efficient exploration and transfer","date":"2020-11-12","arxiv_id":"2011.06335","repositories_listed":0,"syntology":null},{"url":null,"slug":"imposing-robust-structured-control-constraint","title":"Imposing Robust Structured Control Constraint on Reinforcement Learning of Linear Quadratic Regulator","date":"2020-11-12","arxiv_id":"2011.07011","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-reinforcement-learning-for","title":"Self-supervised reinforcement learning for speaker localisation with the iCub humanoid robot","date":"2020-11-12","arxiv_id":"2011.06544","repositories_listed":0,"syntology":null},{"url":null,"slug":"steady-state-analysis-of-episodic-1","title":"Steady State Analysis of Episodic Reinforcement Learning","date":"2020-11-12","arxiv_id":"2011.06631","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-primal-approach-to-constrained-policy-1","title":"CRPO: A New Approach for Safe Reinforcement Learning with Convergence Guarantee","date":"2020-11-11","arxiv_id":"2011.05869","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-neural-architectures-for-recommender","title":"Adaptive Neural Architectures for Recommender Systems","date":"2020-11-11","arxiv_id":"2012.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviorally-diverse-traffic-simulation-via","title":"Behaviorally Diverse Traffic Simulation via Reinforcement Learning","date":"2020-11-11","arxiv_id":"2011.05741","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-q-learning-leveraging-importance-1","title":"On Using Hamiltonian Monte Carlo Sampling for Reinforcement Learning Problems in High-dimension","date":"2020-11-11","arxiv_id":"2011.05927","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-local-optimization-imposing-structure-on","title":"Non-local Optimization: Imposing Structure on Optimization Problems by Relaxation","date":"2020-11-11","arxiv_id":"2011.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-learning-of-counterfactual-perception","title":"Offline Learning of Counterfactual Predictions for Real-World Robotic Reinforcement Learning","date":"2020-11-11","arxiv_id":"2011.05857","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-optimization-via-enhanced","title":"Proximal Policy Optimization via Enhanced Exploration Efficiency","date":"2020-11-11","arxiv_id":"2011.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-time-dependent","title":"Reinforcement Learning with Time-dependent Goals for Robotic Musicians","date":"2020-11-11","arxiv_id":"2011.05715","repositories_listed":0,"syntology":null},{"url":null,"slug":"dirichlet-policies-for-reinforced-factor","title":"Dirichlet policies for reinforced factor portfolios","date":"2020-11-10","arxiv_id":"2011.05381","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-relay-selection-and-power-allocation","title":"Hierarchical Reinforcement Learning for Relay Selection and Power Optimization in Two-Hop Cooperative Relay Network","date":"2020-11-10","arxiv_id":"2011.04891","repositories_listed":0,"syntology":null},{"url":null,"slug":"kinematics-guided-reinforcement-learning-for","title":"Kinematics-Guided Reinforcement Learning for Object-Aware 3D Ego-Pose Estimation","date":"2020-11-10","arxiv_id":"2011.04837","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-from","title":"Model-based Reinforcement Learning from Signal Temporal Logic Specifications","date":"2020-11-10","arxiv_id":"2011.04950","repositories_listed":0,"syntology":null},{"url":null,"slug":"perturbation-based-exploration-methods-in","title":"Perturbation-based exploration methods in deep reinforcement learning","date":"2020-11-10","arxiv_id":"2011.05446","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-bounds-for-two-timescale","title":"Sample Complexity Bounds for Two Timescale Value-based Reinforcement Learning Algorithms","date":"2020-11-10","arxiv_id":"2011.05053","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-adversary-emulation-for-cyber","title":"Automated Adversary Emulation for Cyber-Physical Systems via Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04635","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-planning-at-urban-intersections","title":"Behavior Planning at Urban Intersections through Hierarchical Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04697","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-exploration-and-general-function","title":"On Function Approximation in Reinforcement Learning: Optimism in the Face of Large State Spaces","date":"2020-11-09","arxiv_id":"2011.04622","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-of-applying-deep-reinforcement","title":"Challenges of Applying Deep Reinforcement Learning in Dynamic Dispatching","date":"2020-11-09","arxiv_id":"2011.05570","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-propositional-logic-based-decision","title":"Combining Propositional Logic Based Decision Diagrams with Decision Making in Urban Systems","date":"2020-11-09","arxiv_id":"2011.04405","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-navigation-in","title":"Deep Reinforcement Learning for Navigation in AAA Video Games","date":"2020-11-09","arxiv_id":"2011.04764","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-ran","title":"Deep reinforcement learning for RAN optimization and control","date":"2020-11-09","arxiv_id":"2011.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-compose-hierarchical-object","title":"Learning to Compose Hierarchical Object-Centric Controllers for Robotic Manipulation","date":"2020-11-09","arxiv_id":"2011.04627","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-joint","title":"Multi-Agent Reinforcement Learning for Channel Assignment and Power Allocation in Platoon-Based C-V2X Systems","date":"2020-11-09","arxiv_id":"2011.04555","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-age-of-information-through-aerial","title":"Optimizing Age of Information Through Aerial Reconfigurable Intelligent Surfaces: A Deep Reinforcement Learning Approach","date":"2020-11-09","arxiv_id":"2011.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-autonomous-driving","title":"Reinforcement Learning for Autonomous Driving with Latent State Inference and Spatial-Temporal Relationships","date":"2020-11-09","arxiv_id":"2011.04251","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-trajectory-planning-using-reinforcement","title":"Safe Trajectory Planning Using Reinforcement Learning for Self Driving","date":"2020-11-09","arxiv_id":"2011.04702","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-market-power-using-deep","title":"Exploring market power using deep reinforcement learning for intelligent bidding strategies","date":"2020-11-08","arxiv_id":"2011.04079","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-role-of-planning-in-model-based-deep-1","title":"On the role of planning in model-based deep reinforcement learning","date":"2020-11-08","arxiv_id":"2011.04021","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-sparse-reinforcement-learning","title":"Online Sparse Reinforcement Learning","date":"2020-11-08","arxiv_id":"2011.04018","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-assignment-problem","title":"Reinforcement Learning for Assignment problem","date":"2020-11-08","arxiv_id":"2011.03909","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-off-policy-evaluation-for","title":"Reliable Off-policy Evaluation for Reinforcement Learning","date":"2020-11-08","arxiv_id":"2011.04102","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-feature-selection-makes-batch","title":"Sparse Feature Selection Makes Batch Reinforcement Learning More Sample Efficient","date":"2020-11-08","arxiv_id":"2011.04019","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-activation-function-for-machine","title":"Universal Activation Function For Machine Learning","date":"2020-11-07","arxiv_id":"2011.03842","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-prediction-on-self-driving-cars-a","title":"Motion Prediction on Self-driving Cars: A Review","date":"2020-11-06","arxiv_id":"2011.03635","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-in","title":"Sample-efficient Reinforcement Learning in Robotic Table Tennis","date":"2020-11-06","arxiv_id":"2011.03275","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-and-multi-agent-deep-reinforcement","title":"Single and Multi-Agent Deep Reinforcement Learning for AI-Enabled Wireless Networks: A Tutorial","date":"2020-11-06","arxiv_id":"2011.03615","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-value-equivalence-principle-for-model","title":"The Value Equivalence Principle for Model-Based Reinforcement Learning","date":"2020-11-06","arxiv_id":"2011.03506","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hysteretic-q-learning-coordination","title":"A Hysteretic Q-learning Coordination Framework for Emerging Mobility Systems in Smart Cities","date":"2020-11-05","arxiv_id":"2011.03137","repositories_listed":0,"syntology":null},{"url":null,"slug":"lbgp-learning-based-goal-planning-for","title":"LBGP: Learning Based Goal Planning for Autonomous Following in Front","date":"2020-11-05","arxiv_id":"2011.03125","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-utilize-shaping-rewards-a-new","title":"Learning to Utilize Shaping Rewards: A New Approach of Reward Shaping","date":"2020-11-05","arxiv_id":"2011.02669","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-optical-tweezers-with-deep","title":"Playing optical tweezers with deep reinforcement learning: in virtual, physical and augmented environments","date":"2020-11-05","arxiv_id":"2011.04424","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-inverse-deep-reinforcement","title":"Generative Inverse Deep Reinforcement Learning for Online Recommendation","date":"2020-11-04","arxiv_id":"2011.02248","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-human-feedback-challenges-for","title":"Offline Reinforcement Learning from Human Feedback in Real-World Sequence-to-Sequence Tasks","date":"2020-11-04","arxiv_id":"2011.02511","repositories_listed":0,"syntology":null},{"url":null,"slug":"mbvi-model-based-value-initialization-for","title":"Optimal Control-Based Baseline for Guided Exploration in Policy Gradient Methods","date":"2020-11-04","arxiv_id":"2011.02073","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-dynamic-1","title":"Deep Reinforcement Learning Based Dynamic Route Planning for Minimizing Travel Time","date":"2020-11-03","arxiv_id":"2011.01771","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-physics-models-for-real-world","title":"Differentiable Physics Models for Real-world Offline Model-based Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.01734","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-for-3","title":"Distributional Reinforcement Learning for mmWave Communications with Intelligent Reflectors on a UAV","date":"2020-11-03","arxiv_id":"2011.01840","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-observer-based-inverse-reinforcement","title":"Online Observer-Based Inverse Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.02057","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-variant-of-the-wang-foster-kakade-lower","title":"A Variant of the Wang-Foster-Kakade Lower Bound for the Discounted Setting","date":"2020-11-02","arxiv_id":"2011.01075","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-heterogeneous-deep-reinforcement","title":"Cooperative Heterogeneous Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.00791","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-reinforcement-learning-with-incremental","title":"Fast Reinforcement Learning with Incremental Gaussian Mixture Models","date":"2020-11-02","arxiv_id":"2011.00702","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-rivalry-in-reinforcement","title":"Incorporating Rivalry in Reinforcement Learning for a Competitive Game","date":"2020-11-02","arxiv_id":"2011.01337","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-task-selection-for-meta","title":"Information-theoretic Task Selection for Meta-Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpreting-graph-drawing-with-multi-agent","title":"Interpreting Graph Drawing with Multi-Agent Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.00748","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearl-non-explicit-action-reinforcement","title":"NEARL: Non-Explicit Action Reinforcement Learning for Robotic Control","date":"2020-11-02","arxiv_id":"2011.01046","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-structured-control","title":"Reinforcement Learning of Structured Control for Linear Systems with Unknown State Matrix","date":"2020-11-02","arxiv_id":"2011.01128","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-efficient-active","title":"Reinforcement Learning with Efficient Active Feature Acquisition","date":"2020-11-02","arxiv_id":"2011.00825","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-using","title":"Sample-efficient reinforcement learning using deep Gaussian processes","date":"2020-11-02","arxiv_id":"2011.01226","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-rewards-for-reinforcement-learning","title":"Shaping Rewards for Reinforcement Learning with Imperfect Demonstrations using Generative Models","date":"2020-11-02","arxiv_id":"2011.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-multi-hop-relation-reasoning-over","title":"Few-Shot Multi-Hop Relation Reasoning over Knowledge Bases","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"production-based-cognitive-models-as-a-test","title":"Production-based Cognitive Models as a Test Suite for Reinforcement Learning Algorithms","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-imbalanced","title":"Reinforcement Learning with Imbalanced Dataset for Data-to-Text Medical Report Generation","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"task-completion-dialogue-policy-learning-via","title":"Task-Completion Dialogue Policy Learning via Monte Carlo Tree Search with Dueling Network","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-preserving-synthetic-news-generation-an","title":"Topic-Preserving Synthetic News Generation: An Adversarial Deep Reinforcement Learning Approach","date":"2020-10-30","arxiv_id":"2010.16324","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstract-value-iteration-for-hierarchical","title":"Abstract Value Iteration for Hierarchical Reinforcement Learning","date":"2020-10-29","arxiv_id":"2010.15638","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-variables-from-reinforcement-learning","title":"Reinforcement Learning of Causal Variables Using Mediation Analysis","date":"2020-10-29","arxiv_id":"2010.15745","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-do-offline-measures-for-exploration-in","title":"How do Offline Measures for Exploration in Reinforcement Learning behave?","date":"2020-10-29","arxiv_id":"2010.15533","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-versus-machine-attention-in-deep","title":"Machine versus Human Attention in Deep Reinforcement Learning Tasks","date":"2020-10-29","arxiv_id":"2010.15942","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-personalized-discretionary-lane","title":"Learning Personalized Discretionary Lane-Change Initiation for Fully Autonomous Driving Based on Reinforcement Learning","date":"2020-10-29","arxiv_id":"2010.15372","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepfoldit-a-deep-reinforcement-learning","title":"DeepFoldit -- A Deep Reinforcement Learning Neural Network Folding Proteins","date":"2020-10-28","arxiv_id":"2011.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-interpretable-approximations-to","title":"Designing Interpretable Approximations to Deep Reinforcement Learning","date":"2020-10-28","arxiv_id":"2010.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-unknot","title":"Learning to Unknot","date":"2020-10-28","arxiv_id":"2010.16263","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sparse-reward-1","title":"Reinforcement Learning for Sparse-Reward Object-Interaction Tasks in a First-person Simulated 3D Environment","date":"2020-10-28","arxiv_id":"2010.15195","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-pathologies-of-approximate","title":"Understanding the Pathologies of Approximate Policy Evaluation when Combined with Greedification in Reinforcement Learning","date":"2020-10-28","arxiv_id":"2010.15268","repositories_listed":0,"syntology":null},{"url":null,"slug":"affordance-as-general-value-function-a","title":"Affordance as general value function: A computational model","date":"2020-10-27","arxiv_id":"2010.14289","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-reinforcement-learning-with-a","title":"Batch Reinforcement Learning with a Nonparametric Off-Policy Policy Gradient","date":"2020-10-27","arxiv_id":"2010.14771","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-priors-for-efficient-reinforcement","title":"Behavior Priors for Efficient Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14274","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-reinforcement-learning-for-continuous","title":"Can Reinforcement Learning for Continuous Control Generalize Across Physics Engines?","date":"2020-10-27","arxiv_id":"2010.14444","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-safety-critics-for-exploration-1","title":"Conservative Safety Critics for Exploration","date":"2020-10-27","arxiv_id":"2010.14497","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-time-reduction-using-warm-start","title":"Learning Time Reduction Using Warm Start Methods for a Reinforcement Learning Based Supervisory Control in Hybrid Electric Vehicle Applications","date":"2020-10-27","arxiv_id":"2010.14575","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-be-safe-deep-rl-with-a-safety","title":"Learning to be Safe: Deep RL with a Safety Critic","date":"2020-10-27","arxiv_id":"2010.14603","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-safe-policy-learning-for-power","title":"Multi-Agent Safe Policy Learning for Power Management of Networked Microgrids","date":"2020-10-27","arxiv_id":"1907.02091","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-decision-making-for-urban","title":"Behavioral decision-making for urban autonomous driving in the presence of pedestrians using Deep Recurrent Q-Network","date":"2020-10-26","arxiv_id":"2010.13407","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-latent-movements-off-policy","title":"Contextual Latent-Movements Off-Policy Optimization for Robotic Manipulation Skills","date":"2020-10-26","arxiv_id":"2010.13766","repositories_listed":0,"syntology":null},{"url":null,"slug":"forethought-and-hindsight-in-credit","title":"Forethought and Hindsight in Credit Assignment","date":"2020-10-26","arxiv_id":"2010.13685","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-acceleration-reinforcement-learning-for","title":"High Acceleration Reinforcement Learning for Real-World Juggling with Binary Rewards","date":"2020-10-26","arxiv_id":"2010.13483","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-state","title":"Lyapunov-Based Reinforcement Learning State Estimator","date":"2020-10-26","arxiv_id":"2010.13529","repositories_listed":0,"syntology":null},{"url":null,"slug":"opal-offline-primitive-discovery-for-1","title":"OPAL: Offline Primitive Discovery for Accelerating Offline Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairwise-heuristic-sequence-alignment","title":"Pairwise heuristic sequence alignment algorithm based on deep reinforcement learning","date":"2020-10-26","arxiv_id":"2010.13478","repositories_listed":0,"syntology":null},{"url":null,"slug":"track-assignment-detailed-routing-using","title":"Track-Assignment Detailed Routing Using Attention-based Policy Model With Supervision","date":"2020-10-26","arxiv_id":"2010.13702","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualhints-a-visual-lingual-environment-for","title":"VisualHints: A Visual-Lingual Environment for Multimodal Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-federated-learning-and-digital-twin","title":"Adaptive Federated Learning and Digital Twin for Industrial Internet of Things","date":"2020-10-25","arxiv_id":"2010.13058","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reinforcement-learning-by-a-finite","title":"Enhancing reinforcement learning by a finite reward response filter with a case study in intelligent structural control","date":"2020-10-25","arxiv_id":"2010.15597","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-exploration-of-deep","title":"Improving the Exploration of Deep Reinforcement Learning in Continuous Domains using Planning for Policy Search","date":"2020-10-24","arxiv_id":"2010.12974","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-exploration-addressing-dynamics","title":"Planning with Exploration: Addressing Dynamics Bottleneck in Model-based Reinforcement Learning","date":"2020-10-24","arxiv_id":"2010.12914","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-worst-case-regret-bounds-for","title":"Improved Worst-Case Regret Bounds for Randomized Least-Squares Value Iteration","date":"2020-10-23","arxiv_id":"2010.12163","repositories_listed":0,"syntology":null}],"record_sha256":"602f449347227d9e46ef2579485db023c2669dfe246d057d4bb1e3edc241ce90","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}