{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/85","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":85,"pages_in_order":132,"rows_per_page":100,"rows":[8401,8500],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/84","next":"/task/reinforcement-learning/papers/86","papers":[{"url":null,"slug":"reward-shaping-with-subgoals-for-social","title":"Reward Shaping with Subgoals for Social Navigation","date":"2021-04-13","arxiv_id":"2104.06410","repositories_listed":0,"syntology":null},{"url":null,"slug":"subgoal-based-reward-shaping-to-improve","title":"Subgoal-based Reward Shaping to Improve Efficiency in Reinforcement Learning","date":"2021-04-13","arxiv_id":"2104.06411","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-controller","title":"Deep Reinforcement Learning Based Controller for Active Heave Compensation","date":"2021-04-12","arxiv_id":"2104.05599","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-reinforcement-learning-for-language","title":"Survey on reinforcement learning for language processing","date":"2021-04-12","arxiv_id":"2104.05565","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-a-control","title":"Inverse Reinforcement Learning: A Control Lyapunov Approach","date":"2021-04-09","arxiv_id":"2104.04483","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reweight-imaginary-transitions","title":"Learning to Reweight Imaginary Transitions for Model-Based Reinforcement Learning","date":"2021-04-09","arxiv_id":"2104.04174","repositories_listed":0,"syntology":null},{"url":null,"slug":"acerac-efficient-reinforcement-learning-in","title":"ACERAC: Efficient reinforcement learning in fine time discretization","date":"2021-04-08","arxiv_id":"2104.04004","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-for-two-time-scale-non","title":"Non-Asymptotic Analysis for Two Time-scale TDC with General Smooth Function Approximation","date":"2021-04-07","arxiv_id":"2104.02836","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-a-disentangled","title":"Reinforcement Learning with a Disentangled Universal Value Function for Item Recommendation","date":"2021-04-07","arxiv_id":"2104.02981","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-visual-attention-and-invariance","title":"Unsupervised Visual Attention and Invariance for Reinforcement Learning","date":"2021-04-07","arxiv_id":"2104.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-robust-nmpc-using-reinforcement","title":"Approximate Robust NMPC using Reinforcement Learning","date":"2021-04-06","arxiv_id":"2104.02743","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-simulation-of-ride-hailing","title":"Data-Driven Simulation of Ride-Hailing Services using Imitation and Reinforcement Learning","date":"2021-04-06","arxiv_id":"2104.02661","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for-3","title":"Distributed Deep Reinforcement Learning for Collaborative Spectrum Sharing","date":"2021-04-06","arxiv_id":"2104.02059","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-extension-of-reinforcement","title":"Progressive extension of reinforcement learning action dimension for asymmetric assembly tasks","date":"2021-04-06","arxiv_id":"2104.04078","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeus-efficiently-localizing-actions-in-videos","title":"Zeus: Efficiently Localizing Actions in Videos using Reinforcement Learning","date":"2021-04-06","arxiv_id":"2104.06142","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-critic-reinforcement-learning","title":"A Dual-Critic Reinforcement Learning Framework for Frame-level Bit Allocation in HEVC/H.265","date":"2021-04-05","arxiv_id":"2104.01735","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-transformers-in-reinforcement-1","title":"Efficient Transformers in Reinforcement Learning using Actor-Learner Distillation","date":"2021-04-04","arxiv_id":"2104.01655","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dynamics-perspective-of-pursuit-evasion","title":"A Dynamics Perspective of Pursuit-Evasion Games of Intelligent Agents with the Ability to Learn","date":"2021-04-03","arxiv_id":"2104.01445","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-powered-irs","title":"Deep Reinforcement Learning Powered IRS-Assisted Downlink NOMA","date":"2021-04-03","arxiv_id":"2104.01414","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-emotional-text-to","title":"Reinforcement Learning for Emotional Text-to-Speech Synthesis with Improved Emotion Discriminability","date":"2021-04-03","arxiv_id":"2104.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-double-deep-q-learning-for-joint","title":"Federated Double Deep Q-learning for Joint Delay and Energy Minimization in IoT networks","date":"2021-04-02","arxiv_id":"2104.11320","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dose-helical-cbct-denoising-by-using","title":"Low Dose Helical CBCT denoising by using domain filtering with deep reinforcement learning","date":"2021-04-02","arxiv_id":"2104.00889","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealio-data-efficient-adversarial-learning","title":"DEALIO: Data-Efficient Adversarial Learning for Imitation from Observation","date":"2021-03-31","arxiv_id":"2104.00163","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-edge-computing-when-lyapunov","title":"Energy Efficient Edge Computing: When Lyapunov Meets Distributed Reinforcement Learning","date":"2021-03-31","arxiv_id":"2103.16985","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-reinforcement-learning-for","title":"Generalized Reinforcement Learning for Building Control using Behavioral Cloning","date":"2021-03-31","arxiv_id":"2104.00123","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-heterogeneous-general-equilibrium","title":"Solving Heterogeneous General Equilibrium Economic Models with Deep Reinforcement Learning","date":"2021-03-31","arxiv_id":"2103.16977","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-policies-for-real-time-control-using","title":"Online Policies for Real-Time Control Using MRAC-RL","date":"2021-03-30","arxiv_id":"2103.16551","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimization-of-1","title":"Reinforcement learning for optimization of variational quantum circuit architectures","date":"2021-03-30","arxiv_id":"2103.16089","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-automated-game-testing-with-deep","title":"Augmenting Automated Game Testing with Deep Reinforcement Learning","date":"2021-03-29","arxiv_id":"2103.15819","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hedging-of-derivatives-using","title":"Deep Hedging of Derivatives Using Reinforcement Learning","date":"2021-03-29","arxiv_id":"2103.16409","repositories_listed":0,"syntology":null},{"url":null,"slug":"laser-learning-a-latent-action-space-for","title":"LASER: Learning a Latent Action Space for Efficient Reinforcement Learning","date":"2021-03-29","arxiv_id":"2103.15793","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-sample-efficiency-and","title":"Measuring Sample Efficiency and Generalization in Reinforcement Learning Benchmarks: NeurIPS 2020 Procgen Benchmark","date":"2021-03-29","arxiv_id":"2103.15332","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-beyond-expectation","title":"Reinforcement Learning Beyond Expectation","date":"2021-03-29","arxiv_id":"2104.00540","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowru-knowledge-reusing-via-knowledge","title":"KnowRU: Knowledge Reusing via Knowledge Distillation in Multi-agent Reinforcement Learning","date":"2021-03-27","arxiv_id":"2103.14891","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-overtaking-in-gran-turismo-sport","title":"Autonomous Overtaking in Gran Turismo Sport Using Curriculum Reinforcement Learning","date":"2021-03-26","arxiv_id":"2103.14666","repositories_listed":0,"syntology":null},{"url":null,"slug":"increasing-the-efficiency-of-policy-learning","title":"Increasing the Efficiency of Policy Learning for Autonomous Vehicles by Multi-Task Representation Learning","date":"2021-03-26","arxiv_id":"2103.14718","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-robust","title":"Reinforcement Learning for Robust Parameterized Locomotion Control of Bipedal Robots","date":"2021-03-26","arxiv_id":"2103.14295","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-meta-reinforcement-learning-approach-to","title":"A Meta-Reinforcement Learning Approach to Process Control","date":"2021-03-25","arxiv_id":"2103.14060","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-program-triggered-reinforcement","title":"Hierarchical Program-Triggered Reinforcement Learning Agents For Automated Driving","date":"2021-03-25","arxiv_id":"2103.13861","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-playtesting-coverage-via-curiosity","title":"Improving Playtesting Coverage via Curiosity Driven Reinforcement Learning Agents","date":"2021-03-25","arxiv_id":"2103.13798","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearly-horizon-free-offline-reinforcement","title":"Nearly Horizon-Free Offline Reinforcement Learning","date":"2021-03-25","arxiv_id":"2103.14077","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-deceiving-reactive","title":"Reinforcement Learning for Deceiving Reactive Jammers in Wireless Networks","date":"2021-03-25","arxiv_id":"2103.14056","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-bounds-and-rademacher-complexity-in","title":"Risk Bounds and Rademacher Complexity in Batch Reinforcement Learning","date":"2021-03-25","arxiv_id":"2103.13883","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-imitation-learning-by-planning","title":"Self-Imitation Learning by Planning","date":"2021-03-25","arxiv_id":"2103.13834","repositories_listed":0,"syntology":null},{"url":null,"slug":"cautiously-optimistic-policy-optimization-and","title":"Cautiously Optimistic Policy Optimization and Exploration with Linear Function Approximation","date":"2021-03-24","arxiv_id":"2103.12923","repositories_listed":0,"syntology":null},{"url":null,"slug":"discriminator-augmented-model-based","title":"Discriminator Augmented Model-Based Reinforcement Learning","date":"2021-03-24","arxiv_id":"2103.12999","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-gradient-convergence-bound-of-federated","title":"The Gradient Convergence Bound of Federated Multi-Agent Reinforcement Learning with Efficient Communication","date":"2021-03-24","arxiv_id":"2103.13026","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exponential-lower-bound-for-linearly","title":"An Exponential Lower Bound for Linearly-Realizable MDPs with Constant Suboptimality Gap","date":"2021-03-23","arxiv_id":"2103.12690","repositories_listed":0,"syntology":null},{"url":null,"slug":"assured-learning-enabled-autonomy-a","title":"Assured Learning-enabled Autonomy: A Metacognitive Reinforcement Learning Framework","date":"2021-03-23","arxiv_id":"2103.12558","repositories_listed":0,"syntology":null},{"url":null,"slug":"hamiltonian-policy-optimization-in","title":"Hamiltonian Policy Optimization in Reinforcement Learning","date":"2021-03-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-6dof-grasping-using-reward","title":"Learning 6DoF Grasping Using Reward-Consistent Demonstration","date":"2021-03-23","arxiv_id":"2103.12321","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-adaptable-policy-via-meta","title":"Meta-Adversarial Inverse Reinforcement Learning for Decision-making Tasks","date":"2021-03-23","arxiv_id":"2103.12694","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-contextual-paraphrase-generation","title":"Unsupervised Contextual Paraphrase Generation using Lexical Control and Reinforcement Learning","date":"2021-03-23","arxiv_id":"2103.12777","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-actor-critic-reinforcement-learning","title":"Improving Actor-Critic Reinforcement Learning via Hamiltonian Monte Carlo Method","date":"2021-03-22","arxiv_id":"2103.12020","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-on-scenario-tree","title":"Reinforcement Learning based on Scenario-tree MPC for ASVs","date":"2021-03-22","arxiv_id":"2103.11949","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-scheduling-based-on-deep-reinforcement","title":"Smart Scheduling based on Deep Reinforcement Learning for Cellular Networks","date":"2021-03-22","arxiv_id":"2103.11542","repositories_listed":0,"syntology":null},{"url":null,"slug":"softmax-with-regularization-better-value","title":"Regularized Softmax Deep Multi-Agent $Q$-Learning","date":"2021-03-22","arxiv_id":"2103.11883","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-pessimism-with-optimism-for-robust","title":"Combining Pessimism with Optimism for Robust and Efficient Model-Based Deep Reinforcement Learning","date":"2021-03-18","arxiv_id":"2103.10369","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-reinforcement-learning-with","title":"Maximum Entropy Reinforcement Learning with Mixture Policies","date":"2021-03-18","arxiv_id":"2103.10176","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-reinforcement-learning-for","title":"Decentralized Reinforcement Learning for Multi-Target Search and Detection by a Team of Drones","date":"2021-03-17","arxiv_id":"2103.09520","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-framework-1","title":"Hierarchical Reinforcement Learning Framework for Stochastic Spaceflight Campaign Design","date":"2021-03-16","arxiv_id":"2103.08981","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-shape-rewards-using-a-game-of","title":"Learning to Shape Rewards using a Game of Two Partners","date":"2021-03-16","arxiv_id":"2103.09159","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-curriculum-reinforcement-learning-for","title":"Goal-constrained Sparse Reinforcement Learning for End-to-End Driving","date":"2021-03-16","arxiv_id":"2103.09189","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-online-reinforcement-learning-1","title":"Accelerating Online Reinforcement Learning via Model-Based Meta-Learning","date":"2021-03-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-drone-racing-with-deep","title":"Autonomous Drone Racing with Deep Reinforcement Learning","date":"2021-03-15","arxiv_id":"2103.08624","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-symbolic-rules-for-interpretable","title":"Learning Symbolic Rules for Interpretable Deep Reinforcement Learning","date":"2021-03-15","arxiv_id":"2103.08228","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-human-kinetics-and-kinematics","title":"Modelling Human Kinetics and Kinematics during Walking using Reinforcement Learning","date":"2021-03-15","arxiv_id":"2103.08125","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-value-of-curriculum","title":"Investigating Value of Curriculum Reinforcement Learning in Autonomous Driving Under Diverse Road and Weather Conditions","date":"2021-03-14","arxiv_id":"2103.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-forex-and-stock-price-prediction","title":"A Survey of Forex and Stock Price Prediction Using Deep Learning","date":"2021-03-13","arxiv_id":"2103.09750","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-computer-approach-to-train-a-machine","title":"Hybrid computer approach to train a machine learning system","date":"2021-03-13","arxiv_id":"2103.07802","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-controller-a-reinforcement-learning","title":"RL-Controller: a reinforcement learning framework for active structural control","date":"2021-03-13","arxiv_id":"2103.07616","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-text-generation-with-global","title":"Constrained Text Generation with Global Guidance -- Case Study on CommonGen","date":"2021-03-12","arxiv_id":"2103.07170","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-approach-to","title":"A Reinforcement Learning Based Approach to Play Calling in Football","date":"2021-03-11","arxiv_id":"2103.06939","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-user-interfaces-with-model-based","title":"Adapting User Interfaces with Model-based Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06807","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-in-consensus-based-multi","title":"Adversarial attacks in consensus-based multi-agent reinforcement learning","date":"2021-03-11","arxiv_id":"2103.06967","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-federated-reinforcement-learning","title":"Multi-Task Federated Reinforcement Learning with Adversaries","date":"2021-03-11","arxiv_id":"2103.06473","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-finite-sample-analysis-of-offline","title":"Sample Complexity of Offline Reinforcement Learning with Deep ReLU Networks","date":"2021-03-11","arxiv_id":"2103.06671","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-reinforcement-learning-for-safe-ran","title":"Symbolic Reinforcement Learning for Safe RAN Control","date":"2021-03-11","arxiv_id":"2103.06602","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-perspective-on-1","title":"An Information-Theoretic Perspective on Credit Assignment in Reinforcement Learning","date":"2021-03-10","arxiv_id":"2103.06224","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-gradient-dqn-reinforcement-learning-a","title":"Full Gradient DQN Reinforcement Learning: A Provably Convergent Scheme","date":"2021-03-10","arxiv_id":"2103.05981","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-context-based-meta-reinforcement","title":"Improving Context-Based Meta-Reinforcement Learning with Self-Supervised Trajectory Contrastive Learning","date":"2021-03-10","arxiv_id":"2103.06386","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-based-1","title":"Multi-Objective Reinforcement Learning based Multi-Microgrid System Optimisation Problem","date":"2021-03-10","arxiv_id":"2103.06380","repositories_listed":0,"syntology":null},{"url":null,"slug":"s4rl-surprisingly-simple-self-supervision-for","title":"S4RL: Surprisingly Simple Self-Supervision for Offline Reinforcement Learning","date":"2021-03-10","arxiv_id":"2103.06326","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-for-reinforcement-learning-in","title":"Challenges for Reinforcement Learning in Healthcare","date":"2021-03-09","arxiv_id":"2103.05612","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-impact-time-guidance-a-learning","title":"A Learning-Based Computational Impact Time Guidance","date":"2021-03-09","arxiv_id":"2103.05196","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-a-class-of-multiple","title":"Learning to Explore a Class of Multiple Reward-Free Environments","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-infer-unseen-contexts-in-causal","title":"Learning to Infer Unseen Contexts in Causal Contextual Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"less-suboptimal-learning-and-control-in","title":"Less Suboptimal Learning and Control in Variational POMDPs","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"minimum-description-length-skills-for","title":"Minimum Description Length Skills for Accelerated Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pretraining-reward-free-representations-for","title":"Pretraining Reward-Free Representations for Data-Efficient Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"psiphi-learning-reinforcement-learning-with-1","title":"PsiPhi-Learning: Reinforcement Learning with Demonstrations using Successor Features and Inverse Temporal Difference Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"resolving-causal-confusion-in-reinforcement","title":"Resolving Causal Confusion in Reinforcement Learning via Robust Exploration","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"solipsistic-reinforcement-learning","title":"Solipsistic Reinforcement Learning","date":"2021-03-09","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-quantum-policies-for","title":"Parametrized quantum policies for reinforcement learning","date":"2021-03-09","arxiv_id":"2103.05577","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-for-1","title":"Adversarial Reinforcement Learning for Procedural Content Generation","date":"2021-03-08","arxiv_id":"2103.04847","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-models-the","title":"A multi-agent reinforcement learning model of reputation and cooperation in human groups","date":"2021-03-08","arxiv_id":"2103.04982","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-reinforcement-learning-for-2","title":"Distributed Reinforcement Learning for Flexible and Efficient UAV Swarm Control","date":"2021-03-08","arxiv_id":"2103.04666","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-cooperative-multi-agent","title":"Provably Efficient Cooperative Multi-Agent Reinforcement Learning with Function Approximation","date":"2021-03-08","arxiv_id":"2103.04972","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-ride-hailing-vehicle-repositioning","title":"Real-world Ride-hailing Vehicle Repositioning using Deep Reinforcement Learning","date":"2021-03-08","arxiv_id":"2103.04555","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-based-mobile-robotics-obstacle","title":"Vision-Based Mobile Robotics Obstacle Avoidance With Deep Reinforcement Learning","date":"2021-03-08","arxiv_id":"2103.04727","repositories_listed":0,"syntology":null}],"record_sha256":"ea36465b00d7265943c35ff26926e02895e54cfb3dd1ecefaf1ac82526422d1d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}