{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/123","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":123,"pages_in_order":152,"rows_per_page":100,"rows":[12201,12300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/122","next":"/task/reinforcement-learning-1/papers/124","papers":[{"url":null,"slug":"a-reinforcement-learning-application-of","title":"A reinforcement learning application of guided Monte Carlo Tree Search algorithm for beam orientation selection in radiation therapy","date":"2020-04-14","arxiv_id":"2004.06244","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for-1","title":"Actor-Critic Deep Reinforcement Learning for Solving Job Shop Scheduling Problems","date":"2020-04-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"extrapolation-in-gridworld-markov-decision","title":"Extrapolation in Gridworld Markov-Decision Processes","date":"2020-04-14","arxiv_id":"2004.06784","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-to-vibration","title":"Reinforcement Learning Approach to Vibration Compensation for Dynamic Feed Drive Systems","date":"2020-04-14","arxiv_id":"2004.09263","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-2","title":"A Deep Reinforcement Learning Framework for Continuous Intraday Market Bidding","date":"2020-04-13","arxiv_id":"2004.05940","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-non-cooperative-meta-modeling-game-for","title":"A non-cooperative meta-modeling game for automated third-party calibrating, validating, and falsifying constitutive laws with parallelized adversarial attacks","date":"2020-04-13","arxiv_id":"2004.09392","repositories_listed":0,"syntology":null},{"url":null,"slug":"aspect-and-opinion-aware-abstractive-review","title":"Aspect and Opinion Aware Abstractive Review Summarization with Reinforced Hard Typed Decoder","date":"2020-04-13","arxiv_id":"2004.05755","repositories_listed":0,"syntology":null},{"url":null,"slug":"k-spin-hamiltonian-for-quantum-resolvable","title":"K-spin Hamiltonian for quantum-resolvable Markov decision processes","date":"2020-04-13","arxiv_id":"2004.06040","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-while-moving-deep-reinforcement-1","title":"Thinking While Moving: Deep Reinforcement Learning with Concurrent Control","date":"2020-04-13","arxiv_id":"2004.06089","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-reasoning-from","title":"Reinforcement Learning via Reasoning from Demonstration","date":"2020-04-12","arxiv_id":"2004.05512","repositories_listed":0,"syntology":null},{"url":null,"slug":"certified-adversarial-robustness-for-deep-1","title":"Certifiable Robustness to Adversarial State Uncertainty in Deep Reinforcement Learning","date":"2020-04-11","arxiv_id":"2004.06496","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-process-1","title":"Deep Reinforcement Learning for Process Control: A Primer for Beginners","date":"2020-04-11","arxiv_id":"2004.05490","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-gaussian-processes","title":"Reinforcement Learning via Gaussian Processes with Neural Network Dual Kernels","date":"2020-04-10","arxiv_id":"2004.05198","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-drl-another","title":"Deep Reinforcement Learning (DRL): Another Perspective for Unsupervised Wireless Localization","date":"2020-04-09","arxiv_id":"2004.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-using-weak-derivatives-for","title":"Policy Gradient using Weak Derivatives for Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04843","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-the-impact-of-non-stationarity-in","title":"Quantifying the Impact of Non-Stationarity in Reinforcement Learning-Based Traffic Signal Control","date":"2020-04-09","arxiv_id":"2004.04778","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-conceptualising-the-language-game-paradigm","title":"Re-conceptualising the Language Game Paradigm in the Framework of Multi-Agent Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04722","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-anytime-bottom-up-rule-learning","title":"Reinforced Anytime Bottom Up Rule Learning for Knowledge Graph Completion","date":"2020-04-09","arxiv_id":"2004.04412","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-without-domain","title":"Adaptive Stress Testing without Domain Heuristics using Go-Explore","date":"2020-04-08","arxiv_id":"2004.04292","repositories_listed":0,"syntology":null},{"url":null,"slug":"monte-carlo-siamese-policy-on-actor-for","title":"Monte-Carlo Siamese Policy on Actor for Satellite Image Super Resolution","date":"2020-04-08","arxiv_id":"2004.03879","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-management-for-blockchain-enabled","title":"Resource Management for Blockchain-enabled Federated Learning: A Deep Reinforcement Learning Approach","date":"2020-04-08","arxiv_id":"2004.04104","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-approximation-with-markov-noise","title":"Stochastic Approximation with Markov Noise: Analysis and applications in reinforcement learning","date":"2020-04-08","arxiv_id":"2012.00805","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-constrained-model-based-reinforcement","title":"Online Constrained Model-based Reinforcement Learning","date":"2020-04-07","arxiv_id":"2004.03499","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilising-prior-knowledge-for-visual","title":"Optimistic Agent: Accurate Graph-Based Value Estimation for More Successful Visual Navigation","date":"2020-04-07","arxiv_id":"2004.03222","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsic-exploration-as-multi-objective-rl","title":"Intrinsic Exploration as Multi-Objective RL","date":"2020-04-06","arxiv_id":"2004.02380","repositories_listed":0,"syntology":null},{"url":null,"slug":"networked-multi-agent-reinforcement-learning","title":"Networked Multi-Agent Reinforcement Learning with Emergent Communication","date":"2020-04-06","arxiv_id":"2004.02780","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-adaptive-control-for","title":"Technical Report: Adaptive Control for Linearizable Systems Using On-Policy Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02766","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-state-abstraction-for-reinforcement","title":"Uniform State Abstraction For Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02919","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-reinforcement-learning-for","title":"Weakly-Supervised Reinforcement Learning for Controllable Behavior","date":"2020-04-06","arxiv_id":"2004.02860","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-3","title":"Multi-agent Reinforcement Learning for Resource Allocation in IoT networks with Edge Computing","date":"2020-04-05","arxiv_id":"2004.02315","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-multi-task-approach-for-multi-hop","title":"Reinforced Multi-task Approach for Multi-hop Question Generation","date":"2020-04-05","arxiv_id":"2004.02143","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-architectures-sac-tac","title":"Reinforcement Learning Architectures: SAC, TAC, and ESAC","date":"2020-04-05","arxiv_id":"2004.02274","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylistic-dialogue-generation-via-information","title":"Stylistic Dialogue Generation via Information-Guided Reinforcement Learning Strategy","date":"2020-04-05","arxiv_id":"2004.02202","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-ensemble-multi-agent-reinforcement","title":"A Deep Ensemble Multi-Agent Reinforcement Learning Approach for Air Traffic Control","date":"2020-04-03","arxiv_id":"2004.01387","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mixed-integer","title":"Reinforcement Learning for Mixed-Integer Problems Based on MPC","date":"2020-04-03","arxiv_id":"2004.01430","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-adjusted-discounted","title":"Average Reward Adjusted Discounted Reinforcement Learning: Near-Blackwell-Optimal Policies for Real-World Applications","date":"2020-04-02","arxiv_id":"2004.00857","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-motion-planning-with-temporal","title":"Continuous Motion Planning with Temporal Logic Specifications using Deep Neural Networks","date":"2020-04-02","arxiv_id":"2004.02610","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-reinforcement-learning-for","title":"Exploration of Reinforcement Learning for Event Camera using Car-like Robots","date":"2020-04-02","arxiv_id":"2004.00801","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-projection-on","title":"Safe Reinforcement Learning via Projection on a Safe Set: How to Achieve Optimality?","date":"2020-04-02","arxiv_id":"2004.00915","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-driven-representation-for-human-in-the","title":"Value Driven Representation for Human-in-the-Loop Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.01223","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-space-optimization-and","title":"Constrained-Space Optimization and Reinforcement Learning for Complex Tasks","date":"2020-04-01","arxiv_id":"2004.00716","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-multi-agent-reinforcement","title":"Counterfactual Multi-Agent Reinforcement Learning with Graph Convolution Communication","date":"2020-04-01","arxiv_id":"2004.00470","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistically-model-checking-pctl","title":"Statistically Model Checking PCTL Specifications on Markov Decision Processes via Reinforcement Learning","date":"2020-04-01","arxiv_id":"2004.00273","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-rayleigh-benard-convection-via","title":"Controlling Rayleigh-Bénard convection via Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14358","repositories_listed":0,"syntology":null},{"url":null,"slug":"leverage-the-average-an-analysis-of","title":"Leverage the Average: an Analysis of KL Regularization in RL","date":"2020-03-31","arxiv_id":"2003.14089","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimicking-evolution-with-reinforcement","title":"Mimicking Evolution with Reinforcement Learning","date":"2020-03-31","arxiv_id":"2004.00048","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-bidding-strategy-without-exploration","title":"Optimal Bidding Strategy without Exploration in Real-time Bidding","date":"2020-03-31","arxiv_id":"2004.00100","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-table-tennis-with-model-free","title":"Robotic Table Tennis with Model-Free Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14398","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-reference-reinforcement-learning","title":"Model-Reference Reinforcement Learning Control of Autonomous Surface Vehicles with Uncertainties","date":"2020-03-30","arxiv_id":"2003.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-knowledge-transfer-in-multi-agent","title":"Parallel Knowledge Transfer in Multi-Agent Reinforcement Learning","date":"2020-03-29","arxiv_id":"2003.13085","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-autonomous-systems-meet-accuracy-and","title":"When Autonomous Systems Meet Accuracy and Transferability through AI: A Survey","date":"2020-03-29","arxiv_id":"2003.12948","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-medical-triage-from-clinicians-using","title":"Learning medical triage from clinicians using Deep Q-Learning","date":"2020-03-28","arxiv_id":"2003.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributional-analysis-of-sampling-based","title":"A Distributional Analysis of Sampling-Based Reinforcement Learning Algorithms","date":"2020-03-27","arxiv_id":"2003.12239","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reward-poisoning-attacks-against","title":"Adaptive Reward-Poisoning Attacks against Reinforcement Learning","date":"2020-03-27","arxiv_id":"2003.12613","repositories_listed":0,"syntology":null},{"url":null,"slug":"airrl-a-reinforcement-learning-approach-to","title":"AirRL: A Reinforcement Learning Approach to Urban Air Quality Inference","date":"2020-03-27","arxiv_id":"2003.12205","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-better-opioid-antagonists-using-deep","title":"Towards Better Opioid Antagonists Using Deep Reinforcement Learning","date":"2020-03-26","arxiv_id":"2004.04768","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-conditional-neural-movement","title":"ACNMP: Skill Transfer and Task Extrapolation through Learning from Demonstration and Reinforcement Learning via Representation Sharing","date":"2020-03-25","arxiv_id":"2003.11334","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-off-policy-estimation-for-infinite-1","title":"Black-box Off-policy Estimation for Infinite-Horizon Reinforcement Learning","date":"2020-03-24","arxiv_id":"2003.11126","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-2","title":"Distributional Reinforcement Learning with Ensembles","date":"2020-03-24","arxiv_id":"2003.10903","repositories_listed":0,"syntology":null},{"url":null,"slug":"driver-modeling-through-deep-reinforcement","title":"Driver Modeling through Deep Reinforcement Learning and Behavioral Game Theory","date":"2020-03-24","arxiv_id":"2003.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-stochastic-gradient","title":"Finite-Time Analysis of Stochastic Gradient Descent under Markov Randomness","date":"2020-03-24","arxiv_id":"2003.10973","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-compact-reward-for-image-captioning-1","title":"Learning Compact Reward for Image Captioning","date":"2020-03-24","arxiv_id":"2003.10925","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-play-soccer-by-reinforcement-and","title":"Learning to Play Soccer by Reinforcement and Applying Sim-to-Real to Compete in the Real World","date":"2020-03-24","arxiv_id":"2003.11102","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-1","title":"Multi-Agent Reinforcement Learning for Problems with Combined Individual and Team Reward","date":"2020-03-24","arxiv_id":"2003.10598","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-in-regularized-mean-field-games","title":"Q-Learning in Regularized Mean-field Games","date":"2020-03-24","arxiv_id":"2003.12151","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-recent-advancements-in-model-based-deep-1","title":"Importance of using appropriate baselines for evaluation of data-efficiency in deep reinforcement learning for Atari","date":"2020-03-23","arxiv_id":"2003.10181","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-relational-background-knowledge","title":"Incorporating Relational Background Knowledge into Reinforcement Learning via Differentiable Inductive Logic Programming","date":"2020-03-23","arxiv_id":"2003.10386","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-walk-spike-based-reinforcement","title":"Learning to Walk: Spike Based Reinforcement Learning for Hexapod Robot Central Pattern Generation","date":"2020-03-22","arxiv_id":"2003.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-economics-and","title":"Reinforcement Learning in Economics and Finance","date":"2020-03-22","arxiv_id":"2003.10014","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-uav-navigation-a-ddpg-based-deep","title":"Autonomous UAV Navigation: A DDPG-based Deep Reinforcement Learning Approach","date":"2020-03-21","arxiv_id":"2003.10923","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-review-of-deep-reinforcement","title":"Comprehensive Review of Deep Reinforcement Learning Methods and Applications in Economics","date":"2020-03-21","arxiv_id":"2004.01509","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-smooth","title":"Deep Reinforcement Learning with Robust and Smooth Policy","date":"2020-03-21","arxiv_id":"2003.09534","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-reinforcement-learning-for-1","title":"Distributed Reinforcement Learning for Cooperative Multi-Robot Object Manipulation","date":"2020-03-21","arxiv_id":"2003.09540","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-weighted-q","title":"Deep Reinforcement Learning with Weighted Q-Learning","date":"2020-03-20","arxiv_id":"2003.09280","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-sets-for-generalization-in-rl","title":"Deep Sets for Generalization in RL","date":"2020-03-20","arxiv_id":"2003.09443","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-multi-time-scale-constraints-in","title":"Deep Constrained Q-learning","date":"2020-03-20","arxiv_id":"2003.09398","repositories_listed":0,"syntology":null},{"url":null,"slug":"exchangeable-input-representations-for","title":"Exchangeable Input Representations for Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.09022","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-enabled-cooperative","title":"Reinforcement learning enabled cooperative spectrum sensing in cognitive radio networks","date":"2020-03-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cognitive-routing-based-on-deep","title":"Towards Cognitive Routing based on Deep Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.12439","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-socially-acceptable-perturbations","title":"Generating Socially Acceptable Perturbations for Efficient Evaluation of Autonomous Vehicles","date":"2020-03-18","arxiv_id":"2003.08034","repositories_listed":0,"syntology":null},{"url":null,"slug":"placement-optimization-with-deep","title":"Placement Optimization with Deep Reinforcement Learning","date":"2020-03-18","arxiv_id":"2003.08445","repositories_listed":0,"syntology":null},{"url":null,"slug":"viewport-aware-deep-reinforcement-learning","title":"Viewport-Aware Deep Reinforcement Learning Approach for 360$^o$ Video Caching","date":"2020-03-18","arxiv_id":"2003.08473","repositories_listed":0,"syntology":null},{"url":null,"slug":"watch-your-back-backdoor-attacks-in-deep","title":"Stop-and-Go: Exploring Backdoor Attacks on Deep Reinforcement Learning-based Traffic Congestion Control Systems","date":"2020-03-17","arxiv_id":"2003.07859","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-performance-in-reinforcement","title":"Improving Performance in Reinforcement Learning by Breaking Generalization in Neural Networks","date":"2020-03-16","arxiv_id":"2003.07417","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-electricity","title":"Reinforcement Learning for Electricity Network Operation","date":"2020-03-16","arxiv_id":"2003.07339","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperation-without-coordination-hierarchical","title":"Model-based Reinforcement Learning for Decentralized Multiagent Rendezvous","date":"2020-03-15","arxiv_id":"2003.06906","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-framework-for-learning-mean-field","title":"A General Framework for Learning Mean-Field Games","date":"2020-03-13","arxiv_id":"2003.06069","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-medical-treatment-for-sepsis-in","title":"Optimizing Medical Treatment for Sepsis in Intensive Care: from Reinforcement Learning to Pre-Trial Evaluation","date":"2020-03-13","arxiv_id":"2003.06474","repositories_listed":0,"syntology":null},{"url":"/paper/taylor-expansion-policy-optimization","slug":"taylor-expansion-policy-optimization","title":"Taylor Expansion Policy Optimization","date":"2020-03-13","arxiv_id":"2003.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-visual-representations-in-embodied","title":"Analyzing Visual Representations in Embodied Navigation Tasks","date":"2020-03-12","arxiv_id":"2003.05993","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-relational-reasoning-in","title":"Heterogeneous Relational Reasoning in Knowledge Graphs with Reinforcement Learning","date":"2020-03-12","arxiv_id":"2003.06050","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-curriculum-learning-for-deep-rl-a","title":"Automatic Curriculum Learning For Deep RL: A Short Survey","date":"2020-03-10","arxiv_id":"2003.04664","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-for-reinforcement","title":"Curriculum Learning for Reinforcement Learning Domains: A Framework and Survey","date":"2020-03-10","arxiv_id":"2003.04960","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobility-management-for-cellular-connected","title":"Mobility Management for Cellular-Connected UAVs: A Learning-Based Approach","date":"2020-03-10","arxiv_id":"2002.01546","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-cost-management-in-smart-meters-using","title":"Privacy-Cost Management in Smart Meters Using Deep Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mitigating","title":"Reinforcement Learning for Mitigating Intermittent Interference in Terahertz Communication Networks","date":"2020-03-10","arxiv_id":"2003.04832","repositories_listed":0,"syntology":null},{"url":null,"slug":"squirl-robust-and-efficient-learning-from","title":"SQUIRL: Robust and Efficient Learning from Video Demonstration of Long-Horizon Robotic Manipulation Tasks","date":"2020-03-10","arxiv_id":"2003.04956","repositories_listed":0,"syntology":null},{"url":"/paper/the-minerl-competition-on-sample-efficient-1","slug":"the-minerl-competition-on-sample-efficient-1","title":"Retrospective Analysis of the 2019 MineRL Competition on Sample Efficient Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.05012","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-minerl-competition-on-sample-efficient-1#ran","syntology_url":"https://syntology.ai/paper/2003.05012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05012"}},"official":null}},{"url":null,"slug":"advancing-renewable-electricity-consumption","title":"Advancing Renewable Electricity Consumption With Reinforcement Learning","date":"2020-03-09","arxiv_id":"2003.04310","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-ai-interaction-loop-training-new","title":"Human AI interaction loop training: New approach for interactive reinforcement learning","date":"2020-03-09","arxiv_id":"2003.04203","repositories_listed":0,"syntology":null}],"record_sha256":"c087468af925fb4aa80253fbbcb1bf2c3b72d1353c6c1549040962f656341ef5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}