{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/95","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":95,"pages_in_order":132,"rows_per_page":100,"rows":[9401,9500],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/94","next":"/task/reinforcement-learning/papers/96","papers":[{"url":"/paper/a-game-theoretic-framework-for-model-based","slug":"a-game-theoretic-framework-for-model-based","title":"A Game Theoretic Framework for Model Based Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07804","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-game-theoretic-framework-for-model-based#ran","syntology_url":"https://syntology.ai/paper/2004.07804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07804"}},"official":null}},{"url":null,"slug":"data-driven-robust-control-using","title":"Data-Driven Robust Control Using Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07690","repositories_listed":0,"syntology":null},{"url":null,"slug":"order-matters-generating-progressive","title":"Order Matters: Generating Progressive Explanations for Planning Tasks in Human-Robot Teaming","date":"2020-04-16","arxiv_id":"2004.07822","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-safety-critical","title":"Reinforcement Learning for Safety-Critical Control under Model Uncertainty, using Control Lyapunov Functions and Control Barrier Functions","date":"2020-04-16","arxiv_id":"2004.07584","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionspotter-deep-reinforcement-learning","title":"ActionSpotter: Deep Reinforcement Learning Framework for Temporal Action Spotting in Videos","date":"2020-04-15","arxiv_id":"2004.06971","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapped-model-learning-and-error","title":"Bootstrapped model learning and error correction for planning with uncertainty in model-based RL","date":"2020-04-15","arxiv_id":"2004.07155","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-bandit-anomaly-detection-for-iot","title":"Contextual-Bandit Anomaly Detection for IoT Data in Distributed Hierarchical Edge Computing","date":"2020-04-15","arxiv_id":"2004.06896","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-deep-reinforcement-learning","title":"Extending Deep Reinforcement Learning Frameworks in Cryptocurrency Market Making","date":"2020-04-15","arxiv_id":"2004.06985","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-input-output-linearizing","title":"Improving Input-Output Linearizing Controllers for Bipedal Robots via Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07276","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-user-pairing-and-association-for","title":"Joint User Pairing and Association for Multicell NOMA: A Pointer Network-based Approach","date":"2020-04-15","arxiv_id":"2004.07395","repositories_listed":0,"syntology":null},{"url":null,"slug":"lambert-language-and-action-learning-using","title":"lamBERT: Language and Action Learning Using Multimodal BERT","date":"2020-04-15","arxiv_id":"2004.07093","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-deep-reinforcement-learning-based","title":"Safe deep reinforcement learning-based constrained optimal control scheme for active distribution networks","date":"2020-04-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-demonstration-of-issues-with-value-based","title":"A Demonstration of Issues with Value-Based Multiobjective Reinforcement Learning Under Stochastic State Transitions","date":"2020-04-14","arxiv_id":"2004.06277","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-application-of","title":"A reinforcement learning application of guided Monte Carlo Tree Search algorithm for beam orientation selection in radiation therapy","date":"2020-04-14","arxiv_id":"2004.06244","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for-1","title":"Actor-Critic Deep Reinforcement Learning for Solving Job Shop Scheduling Problems","date":"2020-04-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-evaluation-of-autonomous-vehicles","title":"Adversarial Evaluation of Autonomous Vehicles in Lane-Change Scenarios","date":"2020-04-14","arxiv_id":"2004.06531","repositories_listed":0,"syntology":null},{"url":null,"slug":"extrapolation-in-gridworld-markov-decision","title":"Extrapolation in Gridworld Markov-Decision Processes","date":"2020-04-14","arxiv_id":"2004.06784","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-to-vibration","title":"Reinforcement Learning Approach to Vibration Compensation for Dynamic Feed Drive Systems","date":"2020-04-14","arxiv_id":"2004.09263","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-2","title":"A Deep Reinforcement Learning Framework for Continuous Intraday Market Bidding","date":"2020-04-13","arxiv_id":"2004.05940","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-non-cooperative-meta-modeling-game-for","title":"A non-cooperative meta-modeling game for automated third-party calibrating, validating, and falsifying constitutive laws with parallelized adversarial attacks","date":"2020-04-13","arxiv_id":"2004.09392","repositories_listed":0,"syntology":null},{"url":null,"slug":"aspect-and-opinion-aware-abstractive-review","title":"Aspect and Opinion Aware Abstractive Review Summarization with Reinforced Hard Typed Decoder","date":"2020-04-13","arxiv_id":"2004.05755","repositories_listed":0,"syntology":null},{"url":null,"slug":"k-spin-hamiltonian-for-quantum-resolvable","title":"K-spin Hamiltonian for quantum-resolvable Markov decision processes","date":"2020-04-13","arxiv_id":"2004.06040","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-curriculum-learning-on-pre-trained","title":"Reinforced Curriculum Learning on Pre-trained Neural Machine Translation Models","date":"2020-04-13","arxiv_id":"2004.05757","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-while-moving-deep-reinforcement-1","title":"Thinking While Moving: Deep Reinforcement Learning with Concurrent Control","date":"2020-04-13","arxiv_id":"2004.06089","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-reasoning-from","title":"Reinforcement Learning via Reasoning from Demonstration","date":"2020-04-12","arxiv_id":"2004.05512","repositories_listed":0,"syntology":null},{"url":null,"slug":"certified-adversarial-robustness-for-deep-1","title":"Certifiable Robustness to Adversarial State Uncertainty in Deep Reinforcement Learning","date":"2020-04-11","arxiv_id":"2004.06496","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-process-1","title":"Deep Reinforcement Learning for Process Control: A Primer for Beginners","date":"2020-04-11","arxiv_id":"2004.05490","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-via-gaussian-processes","title":"Reinforcement Learning via Gaussian Processes with Neural Network Dual Kernels","date":"2020-04-10","arxiv_id":"2004.05198","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-drl-another","title":"Deep Reinforcement Learning (DRL): Another Perspective for Unsupervised Wireless Localization","date":"2020-04-09","arxiv_id":"2004.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-drive-off-road-on-smooth-terrain","title":"Learning to Drive Off Road on Smooth Terrain in Unstructured Environments Using an On-Board Camera and Sparse Aerial Images","date":"2020-04-09","arxiv_id":"2004.04697","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-linear-stochastic-approximation-fine","title":"On Linear Stochastic Approximation: Fine-grained Polyak-Ruppert and Non-Asymptotic Concentration","date":"2020-04-09","arxiv_id":"2004.04719","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-using-weak-derivatives-for","title":"Policy Gradient using Weak Derivatives for Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04843","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-the-impact-of-non-stationarity-in","title":"Quantifying the Impact of Non-Stationarity in Reinforcement Learning-Based Traffic Signal Control","date":"2020-04-09","arxiv_id":"2004.04778","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-conceptualising-the-language-game-paradigm","title":"Re-conceptualising the Language Game Paradigm in the Framework of Multi-Agent Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04722","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-anytime-bottom-up-rule-learning","title":"Reinforced Anytime Bottom Up Rule Learning for Knowledge Graph Completion","date":"2020-04-09","arxiv_id":"2004.04412","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-aware-high-level-decisions-for-automated","title":"Risk-Aware High-level Decisions for Automated Driving at Occluded Intersections with Reinforcement Learning","date":"2020-04-09","arxiv_id":"2004.04450","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-without-domain","title":"Adaptive Stress Testing without Domain Heuristics using Go-Explore","date":"2020-04-08","arxiv_id":"2004.04292","repositories_listed":0,"syntology":null},{"url":null,"slug":"genecai-genetic-evolution-for-acquiring","title":"GeneCAI: Genetic Evolution for Acquiring Compact AI","date":"2020-04-08","arxiv_id":"2004.04249","repositories_listed":0,"syntology":null},{"url":null,"slug":"monte-carlo-siamese-policy-on-actor-for","title":"Monte-Carlo Siamese Policy on Actor for Satellite Image Super Resolution","date":"2020-04-08","arxiv_id":"2004.03879","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-management-for-blockchain-enabled","title":"Resource Management for Blockchain-enabled Federated Learning: A Deep Reinforcement Learning Approach","date":"2020-04-08","arxiv_id":"2004.04104","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-do-you-act-an-empirical-study-to","title":"How Do You Act? An Empirical Study to Understand Behavior of Deep Reinforcement Learning Agents","date":"2020-04-07","arxiv_id":"2004.03237","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-constrained-model-based-reinforcement","title":"Online Constrained Model-based Reinforcement Learning","date":"2020-04-07","arxiv_id":"2004.03499","repositories_listed":0,"syntology":null},{"url":null,"slug":"practical-data-poisoning-attack-against-next","title":"Practical Data Poisoning Attack against Next-Item Recommendation","date":"2020-04-07","arxiv_id":"2004.03728","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilising-prior-knowledge-for-visual","title":"Optimistic Agent: Accurate Graph-Based Value Estimation for More Successful Visual Navigation","date":"2020-04-07","arxiv_id":"2004.03222","repositories_listed":0,"syntology":null},{"url":null,"slug":"b-scst-bayesian-self-critical-sequence","title":"B-SCST: Bayesian Self-Critical Sequence Training for Image Captioning","date":"2020-04-06","arxiv_id":"2004.02435","repositories_listed":0,"syntology":null},{"url":null,"slug":"cnn2gate-toward-designing-a-general-framework","title":"CNN2Gate: Toward Designing a General Framework for Implementation of Convolutional Neural Networks on FPGA","date":"2020-04-06","arxiv_id":"2004.04641","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsic-exploration-as-multi-objective-rl","title":"Intrinsic Exploration as Multi-Objective RL","date":"2020-04-06","arxiv_id":"2004.02380","repositories_listed":0,"syntology":null},{"url":null,"slug":"networked-multi-agent-reinforcement-learning","title":"Networked Multi-Agent Reinforcement Learning with Emergent Communication","date":"2020-04-06","arxiv_id":"2004.02780","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-adaptive-control-for","title":"Technical Report: Adaptive Control for Linearizable Systems Using On-Policy Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02766","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-state-abstraction-for-reinforcement","title":"Uniform State Abstraction For Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02919","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-generative-adversarial-nets-on-atari","title":"Using Generative Adversarial Nets on Atari Games for Feature Extraction in Deep Reinforcement Learning","date":"2020-04-06","arxiv_id":"2004.02762","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-reinforcement-learning-for","title":"Weakly-Supervised Reinforcement Learning for Controllable Behavior","date":"2020-04-06","arxiv_id":"2004.02860","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-learning-of-text-adventure-games","title":"Zero-Shot Learning of Text Adventure Games with Sentence-Level Semantics","date":"2020-04-06","arxiv_id":"2004.02986","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-3","title":"Multi-agent Reinforcement Learning for Resource Allocation in IoT networks with Edge Computing","date":"2020-04-05","arxiv_id":"2004.02315","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-multi-task-approach-for-multi-hop","title":"Reinforced Multi-task Approach for Multi-hop Question Generation","date":"2020-04-05","arxiv_id":"2004.02143","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-architectures-sac-tac","title":"Reinforcement Learning Architectures: SAC, TAC, and ESAC","date":"2020-04-05","arxiv_id":"2004.02274","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylistic-dialogue-generation-via-information","title":"Stylistic Dialogue Generation via Information-Guided Reinforcement Learning Strategy","date":"2020-04-05","arxiv_id":"2004.02202","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-ensemble-multi-agent-reinforcement","title":"A Deep Ensemble Multi-Agent Reinforcement Learning Approach for Air Traffic Control","date":"2020-04-03","arxiv_id":"2004.01387","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mixed-integer","title":"Reinforcement Learning for Mixed-Integer Problems Based on MPC","date":"2020-04-03","arxiv_id":"2004.01430","repositories_listed":0,"syntology":null},{"url":null,"slug":"average-reward-adjusted-discounted","title":"Average Reward Adjusted Discounted Reinforcement Learning: Near-Blackwell-Optimal Policies for Real-World Applications","date":"2020-04-02","arxiv_id":"2004.00857","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-motion-planning-with-temporal","title":"Continuous Motion Planning with Temporal Logic Specifications using Deep Neural Networks","date":"2020-04-02","arxiv_id":"2004.02610","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-reinforcement-learning-for","title":"Exploration of Reinforcement Learning for Event Camera using Car-like Robots","date":"2020-04-02","arxiv_id":"2004.00801","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-agile-robotic-locomotion-skills-by","title":"Learning Agile Robotic Locomotion Skills by Imitating Animals","date":"2020-04-02","arxiv_id":"2004.00784","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantification-of-tomographic-patterns","title":"Automated Quantification of CT Patterns Associated with COVID-19 from Chest CT","date":"2020-04-02","arxiv_id":"2004.01279","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-via-projection-on","title":"Safe Reinforcement Learning via Projection on a Safe Set: How to Achieve Optimality?","date":"2020-04-02","arxiv_id":"2004.00915","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-driven-representation-for-human-in-the","title":"Value Driven Representation for Human-in-the-Loop Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.01223","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-challenge-approaching-tetris-link-with","title":"A New Challenge: Approaching Tetris Link with AI","date":"2020-04-01","arxiv_id":"2004.00377","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-space-optimization-and","title":"Constrained-Space Optimization and Reinforcement Learning for Complex Tasks","date":"2020-04-01","arxiv_id":"2004.00716","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-multi-agent-reinforcement","title":"Counterfactual Multi-Agent Reinforcement Learning with Graph Convolution Communication","date":"2020-04-01","arxiv_id":"2004.00470","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-swarm-behavior-in-artificial","title":"Development of swarm behavior in artificial learning agents that adapt to different foraging environments","date":"2020-04-01","arxiv_id":"2004.00552","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistically-model-checking-pctl","title":"Statistically Model Checking PCTL Specifications on Markov Decision Processes via Reinforcement Learning","date":"2020-04-01","arxiv_id":"2004.00273","repositories_listed":0,"syntology":null},{"url":null,"slug":"work-in-progress-temporally-extended","title":"Work in Progress: Temporally Extended Auxiliary Tasks","date":"2020-04-01","arxiv_id":"2004.00600","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-rayleigh-benard-convection-via","title":"Controlling Rayleigh-Bénard convection via Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14358","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-rolling-horizon-evolution-algorithm","title":"Enhanced Rolling Horizon Evolution Algorithm with Opponent Model Learning: Results for the Fighting Game AI Competition","date":"2020-03-31","arxiv_id":"2003.13949","repositories_listed":0,"syntology":null},{"url":null,"slug":"leverage-the-average-an-analysis-of","title":"Leverage the Average: an Analysis of KL Regularization in RL","date":"2020-03-31","arxiv_id":"2003.14089","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimicking-evolution-with-reinforcement","title":"Mimicking Evolution with Reinforcement Learning","date":"2020-03-31","arxiv_id":"2004.00048","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-bidding-strategy-without-exploration","title":"Optimal Bidding Strategy without Exploration in Real-time Bidding","date":"2020-03-31","arxiv_id":"2004.00100","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-table-tennis-with-model-free","title":"Robotic Table Tennis with Model-Free Reinforcement Learning","date":"2020-03-31","arxiv_id":"2003.14398","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-group-sparse-regularization-for","title":"Continual Learning with Node-Importance based Adaptive Group Sparse Regularization","date":"2020-03-30","arxiv_id":"2003.13726","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-reference-reinforcement-learning","title":"Model-Reference Reinforcement Learning Control of Autonomous Surface Vehicles with Uncertainties","date":"2020-03-30","arxiv_id":"2003.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-flows-and-geometric-optimization","title":"Stochastic Flows and Geometric Optimization on the Orthogonal Group","date":"2020-03-30","arxiv_id":"2003.13563","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-and-testing-variable-partitions","title":"Learning and Testing Variable Partitions","date":"2020-03-29","arxiv_id":"2003.12990","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-knowledge-transfer-in-multi-agent","title":"Parallel Knowledge Transfer in Multi-Agent Reinforcement Learning","date":"2020-03-29","arxiv_id":"2003.13085","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-autonomous-systems-meet-accuracy-and","title":"When Autonomous Systems Meet Accuracy and Transferability through AI: A Survey","date":"2020-03-29","arxiv_id":"2003.12948","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-medical-triage-from-clinicians-using","title":"Learning medical triage from clinicians using Deep Q-Learning","date":"2020-03-28","arxiv_id":"2003.12828","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamlined-empirical-bayes-fitting-of-linear","title":"Streamlined Empirical Bayes Fitting of Linear Mixed Models in Mobile Health","date":"2020-03-28","arxiv_id":"2003.12881","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributional-analysis-of-sampling-based","title":"A Distributional Analysis of Sampling-Based Reinforcement Learning Algorithms","date":"2020-03-27","arxiv_id":"2003.12239","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reward-poisoning-attacks-against","title":"Adaptive Reward-Poisoning Attacks against Reinforcement Learning","date":"2020-03-27","arxiv_id":"2003.12613","repositories_listed":0,"syntology":null},{"url":null,"slug":"airrl-a-reinforcement-learning-approach-to","title":"AirRL: A Reinforcement Learning Approach to Urban Air Quality Inference","date":"2020-03-27","arxiv_id":"2003.12205","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-better-opioid-antagonists-using-deep","title":"Towards Better Opioid Antagonists Using Deep Reinforcement Learning","date":"2020-03-26","arxiv_id":"2004.04768","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-conditional-neural-movement","title":"ACNMP: Skill Transfer and Task Extrapolation through Learning from Demonstration and Reinforcement Learning via Representation Sharing","date":"2020-03-25","arxiv_id":"2003.11334","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-ground-truth-an-evaluator","title":"AliExpress Learning-To-Rank: Maximizing Online Model Performance without Going Online","date":"2020-03-25","arxiv_id":"2003.11941","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-recursive-stochastic","title":"Convergence of Recursive Stochastic Algorithms using Wasserstein Divergence","date":"2020-03-25","arxiv_id":"2003.11403","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-off-policy-estimation-for-infinite-1","title":"Black-box Off-policy Estimation for Infinite-Horizon Reinforcement Learning","date":"2020-03-24","arxiv_id":"2003.11126","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-2","title":"Distributional Reinforcement Learning with Ensembles","date":"2020-03-24","arxiv_id":"2003.10903","repositories_listed":0,"syntology":null},{"url":null,"slug":"driver-modeling-through-deep-reinforcement","title":"Driver Modeling through Deep Reinforcement Learning and Behavioral Game Theory","date":"2020-03-24","arxiv_id":"2003.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-stochastic-gradient","title":"Finite-Time Analysis of Stochastic Gradient Descent under Markov Randomness","date":"2020-03-24","arxiv_id":"2003.10973","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-schedule-leasch-a-deep-reinforcement","title":"Learn to Schedule (LEASCH): A Deep reinforcement learning approach for radio resource scheduling in the 5G MAC layer","date":"2020-03-24","arxiv_id":"2003.11003","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-compact-reward-for-image-captioning-1","title":"Learning Compact Reward for Image Captioning","date":"2020-03-24","arxiv_id":"2003.10925","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-play-soccer-by-reinforcement-and","title":"Learning to Play Soccer by Reinforcement and Applying Sim-to-Real to Compete in the Real World","date":"2020-03-24","arxiv_id":"2003.11102","repositories_listed":0,"syntology":null}],"record_sha256":"21f55e8adba763d66a01568a815ef3563a7ad3a795dc5f44ac37628e124f627f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}