{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/91","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":91,"pages_in_order":135,"rows_per_page":100,"rows":[9001,9100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/90","next":"/task/reinforcement-learning-2/papers/92","papers":[{"url":null,"slug":"carl-conditional-value-at-risk-adversarial","title":"ACReL: Adversarial Conditional value-at-risk Reinforcement Learning","date":"2021-09-20","arxiv_id":"2109.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-natural-language-generation-from","title":"Learning Natural Language Generation from Scratch","date":"2021-09-20","arxiv_id":"2109.09371","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-finite-horizon","title":"Reinforcement Learning for Finite-Horizon Restless Multi-Armed Multi-Action Bandits","date":"2021-09-20","arxiv_id":"2109.09855","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-approaches-to-building-collaborative-task","title":"Two Approaches to Building Collaborative, Task-Oriented Dialog Agents through Self-Play","date":"2021-09-20","arxiv_id":"2109.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-behavior-regularized-reinforcement","title":"Dual Behavior Regularized Reinforcement Learning","date":"2021-09-19","arxiv_id":"2109.09037","repositories_listed":0,"syntology":null},{"url":null,"slug":"greedy-unmixing-for-q-learning-in-multi-agent","title":"Greedy UnMixing for Q-Learning in Multi-Agent Reinforcement Learning","date":"2021-09-19","arxiv_id":"2109.09034","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-robotic-reinforcement-learning-by","title":"Lifelong Robotic Reinforcement Learning by Retaining Experiences","date":"2021-09-19","arxiv_id":"2109.09180","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularize-don-t-mix-multi-agent","title":"Regularize! Don't Mix: Multi-Agent Reinforcement Learning without Explicit Centralized Structures","date":"2021-09-19","arxiv_id":"2109.09038","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-offline-reinforcement-learning","title":"Accelerating Offline Reinforcement Learning Application in Real-Time Bidding and Recommendation: Potential Use of Simulation","date":"2021-09-17","arxiv_id":"2109.08331","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordinated-random-access-for-industrial-iot","title":"Coordinated Random Access for Industrial IoT With Correlated Traffic By Reinforcement-Learning","date":"2021-09-17","arxiv_id":"2109.08389","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-global-connectivity-maintenance","title":"Decentralized Global Connectivity Maintenance for Multi-Robot Navigation: A Reinforcement Learning Approach","date":"2021-09-17","arxiv_id":"2109.08536","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-2","title":"Deep Reinforcement Learning Based Multidimensional Resource Management for Energy Harvesting Cognitive NOMA Communications","date":"2021-09-17","arxiv_id":"2109.09503","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-actor-critic-with-integer-actions","title":"Soft Actor-Critic With Integer Actions","date":"2021-09-17","arxiv_id":"2109.08512","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-control-of-quadratic-costs-in-linear","title":"Reinforcement Learning Policies in Continuous-Time Linear Systems","date":"2021-09-16","arxiv_id":"2109.07630","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-and-unification-of-three","title":"Comparison and Unification of Three Regularization Methods in Batch Reinforcement Learning","date":"2021-09-16","arxiv_id":"2109.08134","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-data-sharing-for-multi-task","title":"Conservative Data Sharing for Multi-Task Offline Reinforcement Learning","date":"2021-09-16","arxiv_id":"2109.08128","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-risk-aware-reinforcement-learning","title":"Enabling risk-aware Reinforcement Learning for medical interventions through uncertainty decomposition","date":"2021-09-16","arxiv_id":"2109.07827","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-peers-transfer-reinforcement","title":"Learning from Peers: Deep Transfer Reinforcement Learning for Joint Radio and Cache Resource Allocation in 5G RAN Slicing","date":"2021-09-16","arxiv_id":"2109.07999","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapid-rl-a-reconfigurable-architecture-with","title":"RAPID-RL: A Reconfigurable Architecture with Preemptive-Exits for Efficient Deep-Reinforcement Learning","date":"2021-09-16","arxiv_id":"2109.08231","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-on-encrypted-data","title":"Reinforcement Learning on Encrypted Data","date":"2021-09-16","arxiv_id":"2109.08236","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-a-human-in-the-loop-policy","title":"Convergence of a Human-in-the-Loop Policy-Gradient Algorithm With Eligibility Trace Under Reward, Policy, and Advantage Feedback","date":"2021-09-15","arxiv_id":"2109.07054","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-cycling-of-a-heterogenous-battery","title":"Optimal Cycling of a Heterogenous Battery Bank via Reinforcement Learning","date":"2021-09-15","arxiv_id":"2109.07137","repositories_listed":0,"syntology":null},{"url":null,"slug":"short-quantum-circuits-in-reinforcement","title":"Short Quantum Circuits in Reinforcement Learning Policies for the Vehicle Routing Problem","date":"2021-09-15","arxiv_id":"2109.07498","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-homeostatic-reinforcement-learning","title":"Continuous Homeostatic Reinforcement Learning for Self-Regulated Autonomous Agents","date":"2021-09-14","arxiv_id":"2109.06580","repositories_listed":0,"syntology":null},{"url":null,"slug":"dsdf-an-approach-to-handle-stochastic-agents","title":"DSDF: An approach to handle stochastic agents in collaborative multi-agent reinforcement learning","date":"2021-09-14","arxiv_id":"2109.06609","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-deep-reinforcement-learning-a","title":"Exploration in Deep Reinforcement Learning: From Single-Agent to Multiagent Domain","date":"2021-09-14","arxiv_id":"2109.06668","repositories_listed":0,"syntology":null},{"url":null,"slug":"romax-certifiably-robust-deep-multiagent","title":"ROMAX: Certifiably Robust Deep Multiagent Reinforcement Learning via Convex Relaxation","date":"2021-09-14","arxiv_id":"2109.06795","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-zero-constraint-violation-for","title":"Achieving Zero Constraint Violation for Constrained Reinforcement Learning via Primal-Dual Approach","date":"2021-09-13","arxiv_id":"2109.06332","repositories_listed":0,"syntology":null},{"url":null,"slug":"radars-memory-efficient-reinforcement","title":"RADARS: Memory Efficient Reinforcement Learning Aided Differentiable Neural Architecture Search","date":"2021-09-13","arxiv_id":"2109.05691","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-load-balanced","title":"Reinforcement Learning for Load-balanced Parallel Particle Tracing","date":"2021-09-13","arxiv_id":"2109.05679","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-guarantees-of-fictitious-discount","title":"Theoretical Guarantees of Fictitious Discount Algorithms for Episodic Reinforcement Learning and Global Convergence of Policy Gradient Methods","date":"2021-09-13","arxiv_id":"2109.06362","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-socially-aware-reinforcement-learning-agent","title":"A Socially Aware Reinforcement Learning Agent for The Single Track Road Problem","date":"2021-09-12","arxiv_id":"2109.05486","repositories_listed":0,"syntology":null},{"url":null,"slug":"concave-utility-reinforcement-learning-with","title":"Concave Utility Reinforcement Learning with Zero-Constraint Violations","date":"2021-09-12","arxiv_id":"2109.05439","repositories_listed":0,"syntology":null},{"url":null,"slug":"emvlight-a-decentralized-reinforcement","title":"EMVLight: A Decentralized Reinforcement Learning Framework for Efficient Passage of Emergency Vehicles","date":"2021-09-12","arxiv_id":"2109.05429","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-ensemble-model-based-reinforcement","title":"Federated Ensemble Model-based Reinforcement Learning in Edge Computing","date":"2021-09-12","arxiv_id":"2109.05549","repositories_listed":0,"syntology":null},{"url":null,"slug":"financial-trading-with-feature-preprocessing","title":"Financial Trading with Feature Preprocessing and Recurrent Reinforcement Learning","date":"2021-09-11","arxiv_id":"2109.05283","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-generation-method-for-learning-a-low","title":"Data Generation Method for Learning a Low-dimensional Safe Region in Safe Reinforcement Learning","date":"2021-09-10","arxiv_id":"2109.05077","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-madrl","title":"Multi-agent deep reinforcement learning (MADRL) meets multi-user MIMO systems","date":"2021-09-10","arxiv_id":"2109.04986","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-a-domestic-battery-and-solar","title":"Optimizing a domestic battery and solar photovoltaic system with deep reinforcement learning","date":"2021-09-10","arxiv_id":"2109.05024","repositories_listed":0,"syntology":null},{"url":null,"slug":"projected-state-action-balancing-weights-for","title":"Projected State-action Balancing Weights for Offline Reinforcement Learning","date":"2021-09-10","arxiv_id":"2109.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"incentivizing-an-unknown-crowd","title":"Incentivizing an Unknown Crowd","date":"2021-09-09","arxiv_id":"2109.04226","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-reinforcement-learning-with","title":"Self-supervised Reinforcement Learning with Independently Controllable Subgoals","date":"2021-09-09","arxiv_id":"2109.04150","repositories_listed":0,"syntology":null},{"url":null,"slug":"user-tampering-in-reinforcement-learning","title":"User Tampering in Reinforcement Learning Recommender Systems","date":"2021-09-09","arxiv_id":"2109.04083","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-8","title":"A Deep Reinforcement Learning Approach for Online Parcel Assignment","date":"2021-09-08","arxiv_id":"2109.03467","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-deep-reinforcement-learning-in-1","title":"A Survey of Deep Reinforcement Learning in Recommender Systems: A Systematic Review and Future Directions","date":"2021-09-08","arxiv_id":"2109.03540","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-of-batch-asynchronous-stochastic","title":"Convergence of Batch Asynchronous Stochastic Approximation With Applications to Reinforcement Learning","date":"2021-09-08","arxiv_id":"2109.03445","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-and-adaptive-guidance-and-control","title":"Integrated and Adaptive Guidance and Control for Endoatmospheric Missiles via Reinforcement Learning","date":"2021-09-08","arxiv_id":"2109.03880","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-did-you-learn-that-from-surprising","title":"Membership Inference Attacks Against Temporally Correlated Data in Deep Reinforcement Learning","date":"2021-09-08","arxiv_id":"2109.03975","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-impact-of-mdp-design-for-reinforcement","title":"On the impact of MDP design for Reinforcement Learning agents in Resource Management","date":"2021-09-07","arxiv_id":"2109.03202","repositories_listed":0,"syntology":null},{"url":null,"slug":"delving-into-macro-placement-with","title":"Delving into Macro Placement with Reinforcement Learning","date":"2021-09-06","arxiv_id":"2109.02587","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-global-placement-with-reinforcement","title":"Guiding Global Placement With Reinforcement Learning","date":"2021-09-06","arxiv_id":"2109.02631","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-reward-tweaking-via-conditional","title":"Hindsight Reward Tweaking via Conditional Deep Reinforcement Learning","date":"2021-09-06","arxiv_id":"2109.02332","repositories_listed":0,"syntology":null},{"url":null,"slug":"method-for-making-multi-attribute-decisions","title":"Method for making multi-attribute decisions in wargames by combining intuitionistic fuzzy numbers with reinforcement learning","date":"2021-09-06","arxiv_id":"2109.02354","repositories_listed":0,"syntology":null},{"url":null,"slug":"recommendation-fairness-from-static-to","title":"Recommendation Fairness: From Static to Dynamic","date":"2021-09-05","arxiv_id":"2109.03150","repositories_listed":0,"syntology":null},{"url":null,"slug":"eden-a-unified-environment-framework-for","title":"Eden: A Unified Environment Framework for Booming Reinforcement Learning Algorithms","date":"2021-09-04","arxiv_id":"2109.01768","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-complexity-of-computing-markov-perfect","title":"On the Complexity of Computing Markov Perfect Equilibrium in General-Sum Stochastic Games","date":"2021-09-04","arxiv_id":"2109.01795","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-natural-actor-critic","title":"Multi-agent Natural Actor-critic Reinforcement Learning Algorithms","date":"2021-09-03","arxiv_id":"2109.01654","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-model-based-meta-reinforcement","title":"Provably Safe Model-Based Meta Reinforcement Learning: An Abstraction-Based Approach","date":"2021-09-03","arxiv_id":"2109.01255","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-oracle-and-observations-for-the-openai-gym","title":"An Oracle and Observations for the OpenAI Gym / ALE Freeway Environment","date":"2021-09-02","arxiv_id":"2109.01220","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-inverse-reinforcement-learning-2","title":"Multi-Agent Inverse Reinforcement Learning: Suboptimal Demonstrations and Alternative Solution Concepts","date":"2021-09-02","arxiv_id":"2109.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-battery-energy","title":"Reinforcement Learning for Battery Energy Storage Dispatch augmented with Model-based Optimizer","date":"2021-09-02","arxiv_id":"2109.01659","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-exploration-methods-in","title":"A Survey of Exploration Methods in Reinforcement Learning","date":"2021-09-01","arxiv_id":"2109.00157","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-quantum-reinforcement-learning","title":"Variational Quantum Reinforcement Learning via Evolutionary Optimization","date":"2021-09-01","arxiv_id":"2109.00540","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-deception-into-cyberbattlesim","title":"Incorporating Deception into CyberBattleSim for Autonomous Defense","date":"2021-08-31","arxiv_id":"2108.13980","repositories_listed":0,"syntology":null},{"url":null,"slug":"informing-autonomous-deception-systems-with","title":"Informing Autonomous Deception Systems with Cyber Expert Performance Data","date":"2021-08-31","arxiv_id":"2109.00066","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-perturbation-adversarial-training","title":"Adaptive perturbation adversarial training: based on reinforcement learning","date":"2021-08-30","arxiv_id":"2108.13239","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-vulnerabilities-of-deep-neural","title":"Investigating Vulnerabilities of Deep Neural Policies","date":"2021-08-30","arxiv_id":"2108.13093","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-meta-representations-for-agents-in","title":"Learning Meta Representations for Agents in Multi-Agent Reinforcement Learning","date":"2021-08-30","arxiv_id":"2108.12988","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-policy-efficient-reduction-approach-to","title":"A Policy Efficient Reduction Approach to Convex Constrained Deep Reinforcement Learning","date":"2021-08-29","arxiv_id":"2108.12916","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-sparse-black-box","title":"Reinforcement Learning Based Sparse Black-box Adversarial Attack on Video Recognition Models","date":"2021-08-29","arxiv_id":"2108.13872","repositories_listed":0,"syntology":null},{"url":null,"slug":"influence-based-reinforcement-learning-for","title":"Influence-Based Reinforcement Learning for Intrinsically-Motivated Agents","date":"2021-08-28","arxiv_id":"2108.12581","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-wireless-2","title":"Deep Reinforcement Learning for Wireless Resource Allocation Using Buffer State Information","date":"2021-08-27","arxiv_id":"2108.12198","repositories_listed":0,"syntology":null},{"url":null,"slug":"wad-a-deep-reinforcement-learning-agent-for","title":"WAD: A Deep Reinforcement Learning Agent for Urban Autonomous Driving","date":"2021-08-27","arxiv_id":"2108.12134","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dynamic-band","title":"Deep Reinforcement Learning for Dynamic Band Switch in Cellular-Connected UAV","date":"2021-08-26","arxiv_id":"2108.12054","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-techniques","title":"Federated Reinforcement Learning: Techniques, Applications, and Open Challenges","date":"2021-08-26","arxiv_id":"2108.11887","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-chance-constrained-reinforcement","title":"Model-based Chance-Constrained Reinforcement Learning via Separated Proportional-Integral Lagrangian","date":"2021-08-26","arxiv_id":"2108.11623","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-model-based-reinforcement-learning-for","title":"Robust Model-based Reinforcement Learning for Autonomous Greenhouse Control","date":"2021-08-26","arxiv_id":"2108.11645","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversary-agent-reinforcement-learning-for","title":"Adversary agent reinforcement learning for pursuit-evasion","date":"2021-08-25","arxiv_id":"2108.11010","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-computer","title":"Deep Reinforcement Learning in Computer Vision: A Comprehensive Survey","date":"2021-08-25","arxiv_id":"2108.11510","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-aware-model-initialization-for","title":"Entropy-Aware Model Initialization for Effective Exploration in Deep Reinforcement Learning","date":"2021-08-24","arxiv_id":"2108.10533","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-optimizing-adaptive-optics-control-with-1","title":"Self-optimizing adaptive optics control with Reinforcement Learning for high-contrast imaging","date":"2021-08-24","arxiv_id":"2108.11332","repositories_listed":0,"syntology":null},{"url":null,"slug":"collect-infer-a-fresh-look-at-data-efficient","title":"Collect & Infer -- a fresh look at data-efficient Reinforcement Learning","date":"2021-08-23","arxiv_id":"2108.10273","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-grid-cascading-failure-mitigation-by","title":"Power Grid Cascading Failure Mitigation by Reinforcement Learning","date":"2021-08-23","arxiv_id":"2108.10424","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-boosting-approach-to-reinforcement-learning","title":"A Boosting Approach to Reinforcement Learning","date":"2021-08-22","arxiv_id":"2108.09767","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimicbot-combining-imitation-and","title":"MimicBot: Combining Imitation and Reinforcement Learning to win in Bot Bowl","date":"2021-08-21","arxiv_id":"2108.09478","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-independent-study-of-reinforcement","title":"An Independent Study of Reinforcement Learning and Autonomous Driving","date":"2021-08-20","arxiv_id":"2110.07729","repositories_listed":0,"syntology":null},{"url":null,"slug":"crown-jewels-analysis-using-reinforcement","title":"Crown Jewels Analysis using Reinforcement Learning with Attack Graphs","date":"2021-08-20","arxiv_id":"2108.09358","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-reinforcement-learning-for-broad","title":"Explainable Reinforcement Learning for Broad-XAI: A Conceptual Framework and Survey","date":"2021-08-20","arxiv_id":"2108.09003","repositories_listed":0,"syntology":null},{"url":null,"slug":"plug-and-play-model-based-reinforcement","title":"Plug and Play, Model-Based Reinforcement Learning","date":"2021-08-20","arxiv_id":"2108.08960","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-to-optimize-lifetime","title":"Reinforcement Learning to Optimize Lifetime Value in Cold-Start Recommendation","date":"2021-08-20","arxiv_id":"2108.09141","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-gnss","title":"A Reinforcement Learning Approach for GNSS Spoofing Attack Detection of Autonomous Vehicles","date":"2021-08-19","arxiv_id":"2108.08628","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-localization-utilizing","title":"Cooperative Localization Utilizing Reinforcement Learning for 5G Networks","date":"2021-08-19","arxiv_id":"2108.10222","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-convergence-of-the-ode-limit-for","title":"Global Convergence of the ODE Limit for Online Actor-Critic Algorithms in Reinforcement Learning","date":"2021-08-19","arxiv_id":"2108.08655","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-human-decision-making-with-machine","title":"Improving Human Sequential Decision-Making with Reinforcement Learning","date":"2021-08-19","arxiv_id":"2108.08454","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-benefits-of-actor-critic-methods-for","title":"Provable Benefits of Actor-Critic Methods for Offline Reinforcement Learning","date":"2021-08-19","arxiv_id":"2108.08812","repositories_listed":0,"syntology":null},{"url":null,"slug":"trends-in-neural-architecture-search-towards","title":"Trends in Neural Architecture Search: Towards the Acceleration of Search","date":"2021-08-19","arxiv_id":"2108.08474","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-deep-reinforcement-learning-using","title":"Explainable Deep Reinforcement Learning Using Introspection in a Non-episodic Task","date":"2021-08-18","arxiv_id":"2108.08911","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-placement-of-public-electric-vehicle","title":"Optimal Placement of Public Electric Vehicle Charging Stations Using Deep Reinforcement Learning","date":"2021-08-17","arxiv_id":"2108.07772","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforce-attack-adversarial-attack-against","title":"Reinforce Attack: Adversarial Attack against BERT with Reinforcement Learning","date":"2021-08-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"heterotic-string-model-building-with-monad","title":"Heterotic String Model Building with Monad Bundles and Reinforcement Learning","date":"2021-08-16","arxiv_id":"2108.07316","repositories_listed":0,"syntology":null}],"record_sha256":"18aae8cb218dff7f1ac8b583d9711b4279355089a114b2baa845113a0a00d120","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}