{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/103","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":103,"pages_in_order":135,"rows_per_page":100,"rows":[10201,10300],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/102","next":"/task/reinforcement-learning-2/papers/104","papers":[{"url":null,"slug":"behaviorally-diverse-traffic-simulation-via","title":"Behaviorally Diverse Traffic Simulation via Reinforcement Learning","date":"2020-11-11","arxiv_id":"2011.05741","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-q-learning-leveraging-importance-1","title":"On Using Hamiltonian Monte Carlo Sampling for Reinforcement Learning Problems in High-dimension","date":"2020-11-11","arxiv_id":"2011.05927","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-local-optimization-imposing-structure-on","title":"Non-local Optimization: Imposing Structure on Optimization Problems by Relaxation","date":"2020-11-11","arxiv_id":"2011.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-learning-of-counterfactual-perception","title":"Offline Learning of Counterfactual Predictions for Real-World Robotic Reinforcement Learning","date":"2020-11-11","arxiv_id":"2011.05857","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-optimization-via-enhanced","title":"Proximal Policy Optimization via Enhanced Exploration Efficiency","date":"2020-11-11","arxiv_id":"2011.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-time-dependent","title":"Reinforcement Learning with Time-dependent Goals for Robotic Musicians","date":"2020-11-11","arxiv_id":"2011.05715","repositories_listed":0,"syntology":null},{"url":null,"slug":"dirichlet-policies-for-reinforced-factor","title":"Dirichlet policies for reinforced factor portfolios","date":"2020-11-10","arxiv_id":"2011.05381","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-relay-selection-and-power-allocation","title":"Hierarchical Reinforcement Learning for Relay Selection and Power Optimization in Two-Hop Cooperative Relay Network","date":"2020-11-10","arxiv_id":"2011.04891","repositories_listed":0,"syntology":null},{"url":null,"slug":"kinematics-guided-reinforcement-learning-for","title":"Kinematics-Guided Reinforcement Learning for Object-Aware 3D Ego-Pose Estimation","date":"2020-11-10","arxiv_id":"2011.04837","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-from","title":"Model-based Reinforcement Learning from Signal Temporal Logic Specifications","date":"2020-11-10","arxiv_id":"2011.04950","repositories_listed":0,"syntology":null},{"url":null,"slug":"perturbation-based-exploration-methods-in","title":"Perturbation-based exploration methods in deep reinforcement learning","date":"2020-11-10","arxiv_id":"2011.05446","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-bounds-for-two-timescale","title":"Sample Complexity Bounds for Two Timescale Value-based Reinforcement Learning Algorithms","date":"2020-11-10","arxiv_id":"2011.05053","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-adversary-emulation-for-cyber","title":"Automated Adversary Emulation for Cyber-Physical Systems via Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04635","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-planning-at-urban-intersections","title":"Behavior Planning at Urban Intersections through Hierarchical Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04697","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-of-applying-deep-reinforcement","title":"Challenges of Applying Deep Reinforcement Learning in Dynamic Dispatching","date":"2020-11-09","arxiv_id":"2011.05570","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-propositional-logic-based-decision","title":"Combining Propositional Logic Based Decision Diagrams with Decision Making in Urban Systems","date":"2020-11-09","arxiv_id":"2011.04405","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-navigation-in","title":"Deep Reinforcement Learning for Navigation in AAA Video Games","date":"2020-11-09","arxiv_id":"2011.04764","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-ran","title":"Deep reinforcement learning for RAN optimization and control","date":"2020-11-09","arxiv_id":"2011.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-compose-hierarchical-object","title":"Learning to Compose Hierarchical Object-Centric Controllers for Robotic Manipulation","date":"2020-11-09","arxiv_id":"2011.04627","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-eco","title":"Model-Based Reinforcement Learning for Eco-Driving Control of Electric Vehicles","date":"2020-11-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-autonomous-driving","title":"Reinforcement Learning for Autonomous Driving with Latent State Inference and Spatial-Temporal Relationships","date":"2020-11-09","arxiv_id":"2011.04251","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-trajectory-planning-using-reinforcement","title":"Safe Trajectory Planning Using Reinforcement Learning for Self Driving","date":"2020-11-09","arxiv_id":"2011.04702","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-market-power-using-deep","title":"Exploring market power using deep reinforcement learning for intelligent bidding strategies","date":"2020-11-08","arxiv_id":"2011.04079","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-role-of-planning-in-model-based-deep-1","title":"On the role of planning in model-based deep reinforcement learning","date":"2020-11-08","arxiv_id":"2011.04021","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-sparse-reinforcement-learning","title":"Online Sparse Reinforcement Learning","date":"2020-11-08","arxiv_id":"2011.04018","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-assignment-problem","title":"Reinforcement Learning for Assignment problem","date":"2020-11-08","arxiv_id":"2011.03909","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-off-policy-evaluation-for","title":"Reliable Off-policy Evaluation for Reinforcement Learning","date":"2020-11-08","arxiv_id":"2011.04102","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-feature-selection-makes-batch","title":"Sparse Feature Selection Makes Batch Reinforcement Learning More Sample Efficient","date":"2020-11-08","arxiv_id":"2011.04019","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-prediction-on-self-driving-cars-a","title":"Motion Prediction on Self-driving Cars: A Review","date":"2020-11-06","arxiv_id":"2011.03635","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-in","title":"Sample-efficient Reinforcement Learning in Robotic Table Tennis","date":"2020-11-06","arxiv_id":"2011.03275","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-and-multi-agent-deep-reinforcement","title":"Single and Multi-Agent Deep Reinforcement Learning for AI-Enabled Wireless Networks: A Tutorial","date":"2020-11-06","arxiv_id":"2011.03615","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-value-equivalence-principle-for-model","title":"The Value Equivalence Principle for Model-Based Reinforcement Learning","date":"2020-11-06","arxiv_id":"2011.03506","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hysteretic-q-learning-coordination","title":"A Hysteretic Q-learning Coordination Framework for Emerging Mobility Systems in Smart Cities","date":"2020-11-05","arxiv_id":"2011.03137","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-inverse-deep-reinforcement","title":"Generative Inverse Deep Reinforcement Learning for Online Recommendation","date":"2020-11-04","arxiv_id":"2011.02248","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-human-feedback-challenges-for","title":"Offline Reinforcement Learning from Human Feedback in Real-World Sequence-to-Sequence Tasks","date":"2020-11-04","arxiv_id":"2011.02511","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-dynamic-1","title":"Deep Reinforcement Learning Based Dynamic Route Planning for Minimizing Travel Time","date":"2020-11-03","arxiv_id":"2011.01771","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-physics-models-for-real-world","title":"Differentiable Physics Models for Real-world Offline Model-based Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.01734","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-for-3","title":"Distributional Reinforcement Learning for mmWave Communications with Intelligent Reflectors on a UAV","date":"2020-11-03","arxiv_id":"2011.01840","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-observer-based-inverse-reinforcement","title":"Online Observer-Based Inverse Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.02057","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-variant-of-the-wang-foster-kakade-lower","title":"A Variant of the Wang-Foster-Kakade Lower Bound for the Discounted Setting","date":"2020-11-02","arxiv_id":"2011.01075","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-heterogeneous-deep-reinforcement","title":"Cooperative Heterogeneous Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.00791","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-reinforcement-learning-with-incremental","title":"Fast Reinforcement Learning with Incremental Gaussian Mixture Models","date":"2020-11-02","arxiv_id":"2011.00702","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-rivalry-in-reinforcement","title":"Incorporating Rivalry in Reinforcement Learning for a Competitive Game","date":"2020-11-02","arxiv_id":"2011.01337","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-task-selection-for-meta","title":"Information-theoretic Task Selection for Meta-Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01054","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpreting-graph-drawing-with-multi-agent","title":"Interpreting Graph Drawing with Multi-Agent Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.00748","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearl-non-explicit-action-reinforcement","title":"NEARL: Non-Explicit Action Reinforcement Learning for Robotic Control","date":"2020-11-02","arxiv_id":"2011.01046","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-structured-control","title":"Reinforcement Learning of Structured Control for Linear Systems with Unknown State Matrix","date":"2020-11-02","arxiv_id":"2011.01128","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-efficient-active","title":"Reinforcement Learning with Efficient Active Feature Acquisition","date":"2020-11-02","arxiv_id":"2011.00825","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-using","title":"Sample-efficient reinforcement learning using deep Gaussian processes","date":"2020-11-02","arxiv_id":"2011.01226","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-rewards-for-reinforcement-learning","title":"Shaping Rewards for Reinforcement Learning with Imperfect Demonstrations using Generative Models","date":"2020-11-02","arxiv_id":"2011.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-multi-hop-relation-reasoning-over","title":"Few-Shot Multi-Hop Relation Reasoning over Knowledge Bases","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"production-based-cognitive-models-as-a-test","title":"Production-based Cognitive Models as a Test Suite for Reinforcement Learning Algorithms","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-imbalanced","title":"Reinforcement Learning with Imbalanced Dataset for Data-to-Text Medical Report Generation","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"task-completion-dialogue-policy-learning-via","title":"Task-Completion Dialogue Policy Learning via Monte Carlo Tree Search with Dueling Network","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-preserving-synthetic-news-generation-an","title":"Topic-Preserving Synthetic News Generation: An Adversarial Deep Reinforcement Learning Approach","date":"2020-10-30","arxiv_id":"2010.16324","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstract-value-iteration-for-hierarchical","title":"Abstract Value Iteration for Hierarchical Reinforcement Learning","date":"2020-10-29","arxiv_id":"2010.15638","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-variables-from-reinforcement-learning","title":"Reinforcement Learning of Causal Variables Using Mediation Analysis","date":"2020-10-29","arxiv_id":"2010.15745","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-do-offline-measures-for-exploration-in","title":"How do Offline Measures for Exploration in Reinforcement Learning behave?","date":"2020-10-29","arxiv_id":"2010.15533","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-versus-machine-attention-in-deep","title":"Machine versus Human Attention in Deep Reinforcement Learning Tasks","date":"2020-10-29","arxiv_id":"2010.15942","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepfoldit-a-deep-reinforcement-learning","title":"DeepFoldit -- A Deep Reinforcement Learning Neural Network Folding Proteins","date":"2020-10-28","arxiv_id":"2011.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-interpretable-approximations-to","title":"Designing Interpretable Approximations to Deep Reinforcement Learning","date":"2020-10-28","arxiv_id":"2010.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-reinforcement-learning-with-a","title":"Batch Reinforcement Learning with a Nonparametric Off-Policy Policy Gradient","date":"2020-10-27","arxiv_id":"2010.14771","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-priors-for-efficient-reinforcement","title":"Behavior Priors for Efficient Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14274","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-reinforcement-learning-for-continuous","title":"Can Reinforcement Learning for Continuous Control Generalize Across Physics Engines?","date":"2020-10-27","arxiv_id":"2010.14444","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-decision-making-for-urban","title":"Behavioral decision-making for urban autonomous driving in the presence of pedestrians using Deep Recurrent Q-Network","date":"2020-10-26","arxiv_id":"2010.13407","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-acceleration-reinforcement-learning-for","title":"High Acceleration Reinforcement Learning for Real-World Juggling with Binary Rewards","date":"2020-10-26","arxiv_id":"2010.13483","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-state","title":"Lyapunov-Based Reinforcement Learning State Estimator","date":"2020-10-26","arxiv_id":"2010.13529","repositories_listed":0,"syntology":null},{"url":null,"slug":"opal-offline-primitive-discovery-for-1","title":"OPAL: Offline Primitive Discovery for Accelerating Offline Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairwise-heuristic-sequence-alignment","title":"Pairwise heuristic sequence alignment algorithm based on deep reinforcement learning","date":"2020-10-26","arxiv_id":"2010.13478","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualhints-a-visual-lingual-environment-for","title":"VisualHints: A Visual-Lingual Environment for Multimodal Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-reinforcement-learning-by-a-finite","title":"Enhancing reinforcement learning by a finite reward response filter with a case study in intelligent structural control","date":"2020-10-25","arxiv_id":"2010.15597","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-exploration-of-deep","title":"Improving the Exploration of Deep Reinforcement Learning in Continuous Domains using Planning for Policy Search","date":"2020-10-24","arxiv_id":"2010.12974","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-exploration-addressing-dynamics","title":"Planning with Exploration: Addressing Dynamics Bottleneck in Model-based Reinforcement Learning","date":"2020-10-24","arxiv_id":"2010.12914","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-worst-case-regret-bounds-for","title":"Improved Worst-Case Regret Bounds for Randomized Least-Squares Value Iteration","date":"2020-10-23","arxiv_id":"2010.12163","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-hedging-with-risk-averse-reinforcement","title":"Option Hedging with Risk Averse Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12245","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-inverse-reinforcement-learning-1","title":"Stochastic Inverse Reinforcement Learning","date":"2020-10-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-on-deep-algorithmic","title":"Adversarial Attacks on Deep Algorithmic Trading Policies","date":"2020-10-22","arxiv_id":"2010.11388","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-bounds-of-imitating-policies-and","title":"Error Bounds of Imitating Policies and Environments","date":"2020-10-22","arxiv_id":"2010.11876","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-planner-augmented-reinforcement","title":"Motion Planner Augmented Reinforcement Learning for Robot Manipulation in Obstructed Environments","date":"2020-10-22","arxiv_id":"2010.11940","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimising-stochastic-routing-for-taxi-fleets","title":"Optimising Stochastic Routing for Taxi Fleets with Model Enhanced Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11738","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-with-1","title":"Sample Efficient Reinforcement Learning with REINFORCE","date":"2020-10-22","arxiv_id":"2010.11364","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-are-the-statistical-limits-of-offline-rl","title":"What are the Statistical Limits of Offline RL with Linear Function Approximation?","date":"2020-10-22","arxiv_id":"2010.11895","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-information-asymmetry-in-competitive-multi","title":"On Information Asymmetry in Competitive Multi-Agent Reinforcement Learning: Convergence and Optimality","date":"2020-10-21","arxiv_id":"2010.10901","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-deep-q-networks","title":"Reinforcement learning using Deep Q Networks and Q learning accurately localizes brain tumors on MRI with very small training sets","date":"2020-10-21","arxiv_id":"2010.10763","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-verification-of-model-based-1","title":"Safety Verification of Model Based Reinforcement Learning Controllers","date":"2020-10-21","arxiv_id":"2010.10740","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-lane-merge","title":"Deep Reinforcement Learning in Lane Merge Coordination for Connected Vehicles","date":"2020-10-20","arxiv_id":"2010.10567","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-inference-with-multi-head-automata","title":"Language Inference with Multi-head Automata through Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10141","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-radar-tracking-optimization-for","title":"Multi-Radar Tracking Optimization for Collaborative Combat","date":"2020-10-20","arxiv_id":"2010.11733","repositories_listed":0,"syntology":null},{"url":null,"slug":"negotiating-team-formation-using-deep-1","title":"Negotiating Team Formation Using Deep Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10380","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-of-service-based-radar-resource","title":"Quality of service based radar resource management using deep reinforcement learning","date":"2020-10-20","arxiv_id":"2010.10210","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-constrained-reinforcement-learning-for-1","title":"Robust Constrained Reinforcement Learning for Continuous Control with Model Misspecification","date":"2020-10-20","arxiv_id":"2010.10644","repositories_listed":0,"syntology":null},{"url":null,"slug":"runtime-safety-assurance-using-reinforcement","title":"Runtime Safety Assurance Using Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10618","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-case-for-new-neural-networks-smoothness","title":"A case for new neural networks smoothness constraints","date":"2020-10-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-health","title":"A Reinforcement Learning Approach to Health Aware Control Strategy","date":"2020-10-19","arxiv_id":"2010.09269","repositories_listed":0,"syntology":null},{"url":null,"slug":"chance-constrained-control-with-lexicographic","title":"Chance-Constrained Control with Lexicographic Deep Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09468","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-safety-of-deep-reinforcement","title":"Evaluating the Safety of Deep Reinforcement Learning Models using Semi-Formal Verification","date":"2020-10-19","arxiv_id":"2010.09387","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-with-neural-density-models-1","title":"Imitation with Neural Density Models","date":"2020-10-19","arxiv_id":"2010.09808","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-model-free-reinforcement","title":"Average-reward model-free reinforcement learning: a systematic review and literature mapping","date":"2020-10-18","arxiv_id":"2010.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-inverse-reinforcement-learning","title":"Model-Based Inverse Reinforcement Learning from Visual Demonstrations","date":"2020-10-18","arxiv_id":"2010.09034","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-in-noma","title":"Multi-Agent Reinforcement Learning in NOMA-aided UAV Networks for Cellular Offloading","date":"2020-10-18","arxiv_id":"2010.09094","repositories_listed":0,"syntology":null}],"record_sha256":"9753218fec743af852150f6387b011e0d7cfc313ee636414ef1106026b88b7c2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}