{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/110","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":110,"pages_in_order":135,"rows_per_page":100,"rows":[10901,11000],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/109","next":"/task/reinforcement-learning-2/papers/111","papers":[{"url":null,"slug":"improving-sample-efficiency-and-multi-agent","title":"Improving Sample Efficiency and Multi-Agent Communication in RL-based Train Rescheduling","date":"2020-04-28","arxiv_id":"2004.13439","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-model-selection-in-photonic","title":"Adaptive model selection in photonic reservoir computing by reinforcement learning","date":"2020-04-27","arxiv_id":"2004.12575","repositories_listed":0,"syntology":null},{"url":null,"slug":"age-aware-status-update-control-for-energy","title":"Age-Aware Status Update Control for Energy Harvesting IoT Sensors via Reinforcement Learning","date":"2020-04-27","arxiv_id":"2004.12684","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-learn-heuristics-for-graphical-model","title":"Can We Learn Heuristics For Graphical Model Inference Using Reinforcement Learning?","date":"2020-04-27","arxiv_id":"2005.01508","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ingredients-of-real-world-robotic-1","title":"The Ingredients of Real-World Robotic Reinforcement Learning","date":"2020-04-27","arxiv_id":"2004.12570","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-state-aggregation-approach-for-solving","title":"A State Aggregation Approach for Solving Knapsack Problem with Deep Reinforcement Learning","date":"2020-04-25","arxiv_id":"2004.12117","repositories_listed":0,"syntology":null},{"url":null,"slug":"pbcs-efficient-exploration-and-exploitation","title":"PBCS : Efficient Exploration and Exploitation Using a Synergy between Reinforcement Learning and Motion Planning","date":"2020-04-24","arxiv_id":"2004.11667","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-perception-with-deep","title":"Cooperative Perception with Deep Reinforcement Learning for Connected Vehicles","date":"2020-04-23","arxiv_id":"2004.10927","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-dyna-q-for-mobile-robot-exploration","title":"Guiding Robot Exploration in Reinforcement Learning via Automated Planning","date":"2020-04-23","arxiv_id":"2004.11456","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dialog-policies-from-weak","title":"Learning Dialog Policies from Weak Demonstrations","date":"2020-04-23","arxiv_id":"2004.11054","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoeg-automated-experience-grafting-for-off","title":"AutoEG: Automated Experience Grafting for Off-Policy Deep Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10698","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-and-efficient-long-range-planning-1","title":"Flexible and Efficient Long-Range Planning Through Curious Exploration","date":"2020-04-22","arxiv_id":"2004.10876","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-anomaly-detection-using-inverse","title":"Sequential Anomaly Detection using Inverse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10398","repositories_listed":0,"syntology":null},{"url":null,"slug":"almost-optimal-model-free-reinforcement","title":"Almost Optimal Model-Free Reinforcement Learning via Reference-Advantage Decomposition","date":"2020-04-21","arxiv_id":"2004.10019","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-adaptation-for-end-to-end-vision","title":"Never Stop Learning: The Effectiveness of Fine-Tuning in Robotic Reinforcement Learning","date":"2020-04-21","arxiv_id":"2004.10190","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-to-optimize-the","title":"Reinforcement Learning to Optimize the Logistics Distribution Routes of Unmanned Aerial Vehicle","date":"2020-04-21","arxiv_id":"2004.09864","repositories_listed":0,"syntology":null},{"url":null,"slug":"sibre-self-improvement-based-rewards-for","title":"SIBRE: Self Improvement Based REwards for Adaptive Feedback in Reinforcement Learning","date":"2020-04-21","arxiv_id":"2004.09846","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-routing-track-assignment-detailed","title":"Attention Routing: track-assignment detailed routing using attention-based reinforcement learning","date":"2020-04-20","arxiv_id":"2004.09473","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-learning-and-load-ensemble","title":"Data-Driven Learning and Load Ensemble Control","date":"2020-04-20","arxiv_id":"2004.09675","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-as-reinforcement-applying-principles","title":"Learning as Reinforcement: Applying Principles of Neuroscience for More General Reinforcement Learning Agents","date":"2020-04-20","arxiv_id":"2004.09043","repositories_listed":0,"syntology":null},{"url":null,"slug":"tightening-exploration-in-upper-confidence","title":"Tightening Exploration in Upper Confidence Reinforcement Learning","date":"2020-04-20","arxiv_id":"2004.09656","repositories_listed":0,"syntology":null},{"url":null,"slug":"intention-propagation-for-multi-agent","title":"Variational Policy Propagation for Multi-agent Reinforcement Learning","date":"2020-04-19","arxiv_id":"2004.08883","repositories_listed":0,"syntology":null},{"url":null,"slug":"superkernel-neural-architecture-search-for","title":"Superkernel Neural Architecture Search for Image Denoising","date":"2020-04-19","arxiv_id":"2004.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"macro-action-based-deep-multi-agent","title":"Macro-Action-Based Deep Multi-Agent Reinforcement Learning","date":"2020-04-18","arxiv_id":"2004.08646","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-survival-in-model-based","title":"Modeling Survival in model-based Reinforcement Learning","date":"2020-04-18","arxiv_id":"2004.08648","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-adaptive-reinforcement-learning","title":"Time Adaptive Reinforcement Learning","date":"2020-04-18","arxiv_id":"2004.08600","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-inverse-reinforcement-learning","title":"Approximate Inverse Reinforcement Learning from Vision-based Imitation Learning","date":"2020-04-17","arxiv_id":"2004.08051","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive-1","title":"Deep Reinforcement Learning for Adaptive Learning Systems","date":"2020-04-17","arxiv_id":"2004.08410","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-batch-reinforcement-learning","title":"Goal-conditioned Batch Reinforcement Learning for Rotation Invariant Locomotion","date":"2020-04-17","arxiv_id":"2004.08356","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-guided-deep-reinforcement-learning","title":"Knowledge-guided Deep Reinforcement Learning for Interactive Recommendation","date":"2020-04-17","arxiv_id":"2004.08068","repositories_listed":0,"syntology":null},{"url":null,"slug":"show-us-the-way-learning-to-manage-dialog","title":"Show Us the Way: Learning to Manage Dialog from Demonstrations","date":"2020-04-17","arxiv_id":"2004.08114","repositories_listed":0,"syntology":null},{"url":"/paper/a-game-theoretic-framework-for-model-based","slug":"a-game-theoretic-framework-for-model-based","title":"A Game Theoretic Framework for Model Based Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07804","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-game-theoretic-framework-for-model-based#ran","syntology_url":"https://syntology.ai/paper/2004.07804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07804"}},"official":null}},{"url":null,"slug":"data-driven-robust-control-using","title":"Data-Driven Robust Control Using Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07690","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-safety-critical","title":"Reinforcement Learning for Safety-Critical Control under Model Uncertainty, using Control Lyapunov Functions and Control Barrier Functions","date":"2020-04-16","arxiv_id":"2004.07584","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionspotter-deep-reinforcement-learning","title":"ActionSpotter: Deep Reinforcement Learning Framework for Temporal Action Spotting in Videos","date":"2020-04-15","arxiv_id":"2004.06971","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-deep-reinforcement-learning","title":"Extending Deep Reinforcement Learning Frameworks in Cryptocurrency Market Making","date":"2020-04-15","arxiv_id":"2004.06985","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-input-output-linearizing","title":"Improving Input-Output Linearizing Controllers for Bipedal Robots via Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07276","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-deep-reinforcement-learning-based","title":"Safe deep reinforcement learning-based constrained optimal control scheme for active distribution networks","date":"2020-04-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-demonstration-of-issues-with-value-based","title":"A Demonstration of Issues with Value-Based Multiobjective Reinforcement Learning Under Stochastic State Transitions","date":"2020-04-14","arxiv_id":"2004.06277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-application-of","title":"A reinforcement learning application of guided Monte Carlo Tree Search algorithm for beam orientation selection in radiation therapy","date":"2020-04-14","arxiv_id":"2004.06244","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for-1","title":"Actor-Critic Deep Reinforcement Learning for Solving Job Shop Scheduling Problems","date":"2020-04-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"extrapolation-in-gridworld-markov-decision","title":"Extrapolation in Gridworld Markov-Decision Processes","date":"2020-04-14","arxiv_id":"2004.06784","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-to-vibration","title":"Reinforcement Learning Approach to Vibration Compensation for Dynamic Feed Drive Systems","date":"2020-04-14","arxiv_id":"2004.09263","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-2","title":"A Deep Reinforcement Learning Framework for Continuous Intraday Market Bidding","date":"2020-04-13","arxiv_id":"2004.05940","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-non-cooperative-meta-modeling-game-for","title":"A non-cooperative meta-modeling game for automated third-party calibrating, validating, and falsifying constitutive laws with parallelized adversarial attacks","date":"2020-04-13","arxiv_id":"2004.09392","repositories_listed":0,"syntology":null},{"url":null,"slug":"aspect-and-opinion-aware-abstractive-review","title":"Aspect and Opinion Aware Abstractive Review Summarization with Reinforced Hard Typed Decoder","date":"2020-04-13","arxiv_id":"2004.05755","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-while-moving-deep-reinforcement-1","title":"Thinking While Moving: Deep Reinforcement Learning with Concurrent Control","date":"2020-04-13","arxiv_id":"2004.06089","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-reasoning-from","title":"Reinforcement Learning via Reasoning from Demonstration","date":"2020-04-12","arxiv_id":"2004.05512","repositories_listed":0,"syntology":null},{"url":null,"slug":"certified-adversarial-robustness-for-deep-1","title":"Certifiable Robustness to Adversarial State Uncertainty in Deep Reinforcement Learning","date":"2020-04-11","arxiv_id":"2004.06496","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-process-1","title":"Deep Reinforcement Learning for Process Control: A Primer for Beginners","date":"2020-04-11","arxiv_id":"2004.05490","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-gaussian-processes","title":"Reinforcement Learning via Gaussian Processes with Neural Network Dual Kernels","date":"2020-04-10","arxiv_id":"2004.05198","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-drl-another","title":"Deep Reinforcement Learning (DRL): Another Perspective for Unsupervised Wireless Localization","date":"2020-04-09","arxiv_id":"2004.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-using-weak-derivatives-for","title":"Policy Gradient using Weak Derivatives for Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04843","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-conceptualising-the-language-game-paradigm","title":"Re-conceptualising the Language Game Paradigm in the Framework of Multi-Agent Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04722","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-anytime-bottom-up-rule-learning","title":"Reinforced Anytime Bottom Up Rule Learning for Knowledge Graph Completion","date":"2020-04-09","arxiv_id":"2004.04412","repositories_listed":0,"syntology":null},{"url":null,"slug":"monte-carlo-siamese-policy-on-actor-for","title":"Monte-Carlo Siamese Policy on Actor for Satellite Image Super Resolution","date":"2020-04-08","arxiv_id":"2004.03879","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-management-for-blockchain-enabled","title":"Resource Management for Blockchain-enabled Federated Learning: A Deep Reinforcement Learning Approach","date":"2020-04-08","arxiv_id":"2004.04104","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-approximation-with-markov-noise","title":"Stochastic Approximation with Markov Noise: Analysis and applications in reinforcement learning","date":"2020-04-08","arxiv_id":"2012.00805","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-constrained-model-based-reinforcement","title":"Online Constrained Model-based Reinforcement Learning","date":"2020-04-07","arxiv_id":"2004.03499","repositories_listed":0,"syntology":null},{"url":null,"slug":"networked-multi-agent-reinforcement-learning","title":"Networked Multi-Agent Reinforcement Learning with Emergent Communication","date":"2020-04-06","arxiv_id":"2004.02780","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-adaptive-control-for","title":"Technical Report: Adaptive Control for Linearizable Systems Using On-Policy Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02766","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-state-abstraction-for-reinforcement","title":"Uniform State Abstraction For Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02919","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-reinforcement-learning-for","title":"Weakly-Supervised Reinforcement Learning for Controllable Behavior","date":"2020-04-06","arxiv_id":"2004.02860","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-3","title":"Multi-agent Reinforcement Learning for Resource Allocation in IoT networks with Edge Computing","date":"2020-04-05","arxiv_id":"2004.02315","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-architectures-sac-tac","title":"Reinforcement Learning Architectures: SAC, TAC, and ESAC","date":"2020-04-05","arxiv_id":"2004.02274","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylistic-dialogue-generation-via-information","title":"Stylistic Dialogue Generation via Information-Guided Reinforcement Learning Strategy","date":"2020-04-05","arxiv_id":"2004.02202","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-ensemble-multi-agent-reinforcement","title":"A Deep Ensemble Multi-Agent Reinforcement Learning Approach for Air Traffic Control","date":"2020-04-03","arxiv_id":"2004.01387","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mixed-integer","title":"Reinforcement Learning for Mixed-Integer Problems Based on MPC","date":"2020-04-03","arxiv_id":"2004.01430","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-adjusted-discounted","title":"Average Reward Adjusted Discounted Reinforcement Learning: Near-Blackwell-Optimal Policies for Real-World Applications","date":"2020-04-02","arxiv_id":"2004.00857","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-motion-planning-with-temporal","title":"Continuous Motion Planning with Temporal Logic Specifications using Deep Neural Networks","date":"2020-04-02","arxiv_id":"2004.02610","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-reinforcement-learning-for","title":"Exploration of Reinforcement Learning for Event Camera using Car-like Robots","date":"2020-04-02","arxiv_id":"2004.00801","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-projection-on","title":"Safe Reinforcement Learning via Projection on a Safe Set: How to Achieve Optimality?","date":"2020-04-02","arxiv_id":"2004.00915","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-driven-representation-for-human-in-the","title":"Value Driven Representation for Human-in-the-Loop Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.01223","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-space-optimization-and","title":"Constrained-Space Optimization and Reinforcement Learning for Complex Tasks","date":"2020-04-01","arxiv_id":"2004.00716","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-multi-agent-reinforcement","title":"Counterfactual Multi-Agent Reinforcement Learning with Graph Convolution Communication","date":"2020-04-01","arxiv_id":"2004.00470","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistically-model-checking-pctl","title":"Statistically Model Checking PCTL Specifications on Markov Decision Processes via Reinforcement Learning","date":"2020-04-01","arxiv_id":"2004.00273","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-rayleigh-benard-convection-via","title":"Controlling Rayleigh-Bénard convection via Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14358","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimicking-evolution-with-reinforcement","title":"Mimicking Evolution with Reinforcement Learning","date":"2020-03-31","arxiv_id":"2004.00048","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-bidding-strategy-without-exploration","title":"Optimal Bidding Strategy without Exploration in Real-time Bidding","date":"2020-03-31","arxiv_id":"2004.00100","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-table-tennis-with-model-free","title":"Robotic Table Tennis with Model-Free Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14398","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-reference-reinforcement-learning","title":"Model-Reference Reinforcement Learning Control of Autonomous Surface Vehicles with Uncertainties","date":"2020-03-30","arxiv_id":"2003.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-knowledge-transfer-in-multi-agent","title":"Parallel Knowledge Transfer in Multi-Agent Reinforcement Learning","date":"2020-03-29","arxiv_id":"2003.13085","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-medical-triage-from-clinicians-using","title":"Learning medical triage from clinicians using Deep Q-Learning","date":"2020-03-28","arxiv_id":"2003.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributional-analysis-of-sampling-based","title":"A Distributional Analysis of Sampling-Based Reinforcement Learning Algorithms","date":"2020-03-27","arxiv_id":"2003.12239","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reward-poisoning-attacks-against","title":"Adaptive Reward-Poisoning Attacks against Reinforcement Learning","date":"2020-03-27","arxiv_id":"2003.12613","repositories_listed":0,"syntology":null},{"url":null,"slug":"airrl-a-reinforcement-learning-approach-to","title":"AirRL: A Reinforcement Learning Approach to Urban Air Quality Inference","date":"2020-03-27","arxiv_id":"2003.12205","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-better-opioid-antagonists-using-deep","title":"Towards Better Opioid Antagonists Using Deep Reinforcement Learning","date":"2020-03-26","arxiv_id":"2004.04768","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-off-policy-estimation-for-infinite-1","title":"Black-box Off-policy Estimation for Infinite-Horizon Reinforcement Learning","date":"2020-03-24","arxiv_id":"2003.11126","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-2","title":"Distributional Reinforcement Learning with Ensembles","date":"2020-03-24","arxiv_id":"2003.10903","repositories_listed":0,"syntology":null},{"url":null,"slug":"driver-modeling-through-deep-reinforcement","title":"Driver Modeling through Deep Reinforcement Learning and Behavioral Game Theory","date":"2020-03-24","arxiv_id":"2003.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-stochastic-gradient","title":"Finite-Time Analysis of Stochastic Gradient Descent under Markov Randomness","date":"2020-03-24","arxiv_id":"2003.10973","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-1","title":"Multi-Agent Reinforcement Learning for Problems with Combined Individual and Team Reward","date":"2020-03-24","arxiv_id":"2003.10598","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-in-regularized-mean-field-games","title":"Q-Learning in Regularized Mean-field Games","date":"2020-03-24","arxiv_id":"2003.12151","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-recent-advancements-in-model-based-deep-1","title":"Importance of using appropriate baselines for evaluation of data-efficiency in deep reinforcement learning for Atari","date":"2020-03-23","arxiv_id":"2003.10181","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-relational-background-knowledge","title":"Incorporating Relational Background Knowledge into Reinforcement Learning via Differentiable Inductive Logic Programming","date":"2020-03-23","arxiv_id":"2003.10386","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-walk-spike-based-reinforcement","title":"Learning to Walk: Spike Based Reinforcement Learning for Hexapod Robot Central Pattern Generation","date":"2020-03-22","arxiv_id":"2003.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-economics-and","title":"Reinforcement Learning in Economics and Finance","date":"2020-03-22","arxiv_id":"2003.10014","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-uav-navigation-a-ddpg-based-deep","title":"Autonomous UAV Navigation: A DDPG-based Deep Reinforcement Learning Approach","date":"2020-03-21","arxiv_id":"2003.10923","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-of-deep-reinforcement","title":"Comprehensive Review of Deep Reinforcement Learning Methods and Applications in Economics","date":"2020-03-21","arxiv_id":"2004.01509","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-smooth","title":"Deep Reinforcement Learning with Robust and Smooth Policy","date":"2020-03-21","arxiv_id":"2003.09534","repositories_listed":0,"syntology":null}],"record_sha256":"117b80bd6b52076b2688556133f9bb873a9579c8992b279663f22a91126c85db","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}