{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/113","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":113,"pages_in_order":132,"rows_per_page":100,"rows":[11201,11300],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/112","next":"/task/reinforcement-learning/papers/114","papers":[{"url":null,"slug":"unifying-ensemble-methods-for-q-learning-via","title":"Unifying Ensemble Methods for Q-learning via Social Choice Theory","date":"2019-02-27","arxiv_id":"1902.10646","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-meta-interpretive-learning-outperform","title":"Can Meta-Interpretive Learning outperform Deep Reinforcement Learning of Evaluable Game strategies?","date":"2019-02-26","arxiv_id":"1902.09835","repositories_listed":0,"syntology":null},{"url":null,"slug":"coloring-big-graphs-with-alphagozero","title":"Coloring Big Graphs with AlphaGoZero","date":"2019-02-26","arxiv_id":"1902.10162","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-intentional-backdoors-in-sequential","title":"Design of intentional backdoors in sequential models","date":"2019-02-26","arxiv_id":"1902.09972","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-multi-agent-communication-under","title":"Learning Multi-agent Communication under Limited-bandwidth Restriction for Internet Packet Routing","date":"2019-02-26","arxiv_id":"1903.05561","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-in-hierarchical-reinforcement","title":"Planning in Hierarchical Reinforcement Learning: Guarantees for Using Local Policies","date":"2019-02-26","arxiv_id":"1902.10140","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-termination-critic","title":"The Termination Critic","date":"2019-02-26","arxiv_id":"1902.09996","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-agent-incentives-using-causal","title":"Understanding Agent Incentives using Causal Influence Diagrams. Part I: Single Action Settings","date":"2019-02-26","arxiv_id":"1902.09980","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-ternary-rewards-to-reason-over","title":"Learning When Not to Answer: A Ternary Reward Structure for Reinforcement Learning based Question Answering","date":"2019-02-26","arxiv_id":"1902.10236","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-under","title":"Adversarial Reinforcement Learning under Partial Observability in Autonomous Computer Network Defence","date":"2019-02-25","arxiv_id":"1902.09062","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-modeling-of-dense-and-incomplete","title":"Joint Modeling of Dense and Incomplete Trajectories for Citywide Traffic Volume Inference","date":"2019-02-25","arxiv_id":"1902.09255","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-extreme-hummingbird-maneuvers-on","title":"Learning Extreme Hummingbird Maneuvers on Flapping Wing Robots","date":"2019-02-25","arxiv_id":"1902.09626","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-range-indoor-navigation-with-prm-rl","title":"Long-Range Indoor Navigation with PRM-RL","date":"2019-02-25","arxiv_id":"1902.09458","repositories_listed":0,"syntology":null},{"url":"/paper/making-history-matter-gold-critic-sequence","slug":"making-history-matter-gold-critic-sequence","title":"Making History Matter: History-Advantage Sequence Training for Visual Dialog","date":"2019-02-25","arxiv_id":"1902.09326","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-trigger-continual-state-representation","title":"S-TRIGGER: Continual State Representation Learning via Self-Triggered Generative Replay","date":"2019-02-25","arxiv_id":"1902.09434","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggregating-e-commerce-search-results-from","title":"Aggregating E-commerce Search Results from Heterogeneous Sources via Hierarchical Reinforcement Learning","date":"2019-02-24","arxiv_id":"1902.08882","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-reinforcement","title":"Distributionally Robust Reinforcement Learning","date":"2019-02-23","arxiv_id":"1902.08708","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-memory-for-lifelong-reinforcement","title":"Generative Memory for Lifelong Reinforcement Learning","date":"2019-02-22","arxiv_id":"1902.08349","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-deterministic-policy-with-target-for","title":"Learning Deterministic Policy with Target for Power Control in Wireless Networks","date":"2019-02-21","arxiv_id":"1902.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistics-and-samples-in-distributional","title":"Statistics and Samples in Distributional Reinforcement Learning","date":"2019-02-21","arxiv_id":"1902.08102","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-experience-prioritization","title":"Curiosity-Driven Experience Prioritization via Density Estimation","date":"2019-02-20","arxiv_id":"1902.08039","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-language-to-goals-inverse-reinforcement","title":"From Language to Goals: Inverse Reinforcement Learning for Vision-Based Instruction Following","date":"2019-02-20","arxiv_id":"1902.07742","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-repetition-normalized-adversarial","title":"A novel repetition normalized adversarial reward for headline generation","date":"2019-02-19","arxiv_id":"1902.07110","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-coordination-through-competition","title":"Emergent Coordination Through Competition","date":"2019-02-19","arxiv_id":"1902.07151","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-generalisation-in-continuous","title":"Investigating Generalisation in Continuous Deep Reinforcement Learning","date":"2019-02-19","arxiv_id":"1902.07015","repositories_listed":0,"syntology":null},{"url":null,"slug":"message-dropout-an-efficient-training-method","title":"Message-Dropout: An Efficient Training Method for Multi-Agent Deep Reinforcement Learning","date":"2019-02-18","arxiv_id":"1902.06527","repositories_listed":0,"syntology":null},{"url":null,"slug":"parenting-safe-reinforcement-learning-from","title":"Parenting: Safe Reinforcement Learning from Human Input","date":"2019-02-18","arxiv_id":"1902.06766","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-next-generation-airline-revenue","title":"Autonomous Airline Revenue Management: A Deep Reinforcement Learning Approach to Seat Inventory Control and Overbooking","date":"2019-02-18","arxiv_id":"1902.06824","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-potential-based-reward-shaping-for","title":"A new Potential-Based Reward Shaping for Reinforcement Learning Agent","date":"2019-02-17","arxiv_id":"1902.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-topologies-between-learning","title":"Leveraging Communication Topologies Between Learning Agents in Deep Reinforcement Learning","date":"2019-02-16","arxiv_id":"1902.06740","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-coagent-networks-stochastic","title":"Asynchronous Coagent Networks","date":"2019-02-15","arxiv_id":"1902.05650","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoqb-automl-for-network-quantization-and","title":"AutoQ: Automated Kernel-Wise Neural Network Quantization","date":"2019-02-15","arxiv_id":"1902.05690","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-high-level","title":"Deep Reinforcement Learning Based High-level Driving Behavior Decision-making Model in Heterogeneous Traffic","date":"2019-02-15","arxiv_id":"1902.05772","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-offloading-policies-for-cloud","title":"Network Offloading Policies for Cloud Robotics: a Learning-based Approach","date":"2019-02-15","arxiv_id":"1902.05703","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-in-pomdps-with","title":"Robust Reinforcement Learning in POMDPs with Incomplete and Noisy Observations","date":"2019-02-15","arxiv_id":"1902.05795","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-perception-in-adversarial-scenarios","title":"Active Perception in Adversarial Scenarios using Maximum Entropy Deep Reinforcement Learning","date":"2019-02-14","arxiv_id":"1902.05644","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-and-short-memory-balancing-in-visual-co","title":"Long and Short Memory Balancing in Visual Co-Tracking using Q-Learning","date":"2019-02-14","arxiv_id":"1902.05211","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-actor-critic-in-an-ensemble","title":"Off-Policy Actor-Critic in an Ensemble: Achieving Maximum General Entropy and Effective Environment Exploration in Deep Reinforcement Learning","date":"2019-02-14","arxiv_id":"1902.05551","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reinforcement-learning-using-monte-carlo","title":"Non-Asymptotic Analysis of Monte Carlo Tree Search","date":"2019-02-14","arxiv_id":"1902.05213","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-ua-v-attitude","title":"Reinforcement Learning for UA V Attitude Control","date":"2019-02-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-to-optimize-long-term","title":"Reinforcement Learning to Optimize Long-term User Engagement in Recommender Systems","date":"2019-02-13","arxiv_id":"1902.05570","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneously-learning-vision-and-feature","title":"Simultaneously Learning Vision and Feature-based Control Policies for Real-world Ball-in-a-Cup","date":"2019-02-13","arxiv_id":"1902.04706","repositories_listed":0,"syntology":null},{"url":null,"slug":"actrce-augmenting-experience-via-teachers","title":"ACTRCE: Augmenting Experience via Teacher's Advice For Multi-Goal Reinforcement Learning","date":"2019-02-12","arxiv_id":"1902.04546","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-from-policy","title":"Deep Reinforcement Learning from Policy-Dependent Human Feedback","date":"2019-02-12","arxiv_id":"1902.04257","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-space-reinforcement-learning-for","title":"Latent Space Reinforcement Learning for Steering Angle Prediction","date":"2019-02-11","arxiv_id":"1902.03765","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-whole","title":"Whole-Chain Recommendations","date":"2019-02-11","arxiv_id":"1902.03987","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-dynamics-and-termination-errors","title":"Performance Dynamics and Termination Errors in Reinforcement Learning: A Unifying Perspective","date":"2019-02-11","arxiv_id":"1902.04179","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-reinforcement-learning","title":"Stochastic Reinforcement Learning","date":"2019-02-11","arxiv_id":"1902.04178","repositories_listed":0,"syntology":null},{"url":null,"slug":"wisemove-a-framework-for-safe-deep","title":"WiseMove: A Framework for Safe Deep Reinforcement Learning for Autonomous Driving","date":"2019-02-11","arxiv_id":"1902.04118","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bandit-framework-for-optimal-selection-of","title":"A Bandit Framework for Optimal Selection of Reinforcement Learning Agents","date":"2019-02-10","arxiv_id":"1902.03657","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-neuroevolution-efficiency-by","title":"Improving NeuroEvolution Efficiency by Surrogate Model-based Optimization with Phenotypic Distance Kernels","date":"2019-02-09","arxiv_id":"1902.03419","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with","title":"Distributional reinforcement learning with linear function approximation","date":"2019-02-08","arxiv_id":"1902.03149","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-discount-factor-in","title":"Rethinking the Discount Factor in Reinforcement Learning: A Decision Theoretic Approach","date":"2019-02-08","arxiv_id":"1902.02893","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-impact-of-partner-choice-on","title":"Partner Selection for the Emergence of Cooperation in Multi-Agent Systems Using Reinforcement Learning","date":"2019-02-08","arxiv_id":"1902.03185","repositories_listed":0,"syntology":null},{"url":null,"slug":"klucb-approach-to-copeland-bandits","title":"KLUCB Approach to Copeland Bandits","date":"2019-02-07","arxiv_id":"1902.02778","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaoptimization-on-a-distributed-system-for","title":"Metaoptimization on a Distributed System for Deep Reinforcement Learning","date":"2019-02-07","arxiv_id":"1902.02725","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-search-and-recognition-for-robot-task","title":"Visual search and recognition for robot task execution and monitoring","date":"2019-02-07","arxiv_id":"1902.02870","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-guiding-principle-for-causal-decision","title":"A Guiding Principle for Causal Decision Problems","date":"2019-02-06","arxiv_id":"1902.02279","repositories_listed":0,"syntology":null},{"url":null,"slug":"cesma-centralized-expert-supervises-multi","title":"Decentralized Multi-Agents by Imitation of a Centralized Controller","date":"2019-02-06","arxiv_id":"1902.02311","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-policy-distillation","title":"Distilling Policy Distillation","date":"2019-02-06","arxiv_id":"1902.02186","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-for-sarsa-and-q","title":"Finite-Sample Analysis for SARSA with Linear Function Approximation","date":"2019-02-06","arxiv_id":"1902.02234","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-fictitious-self-play-on-elf-mini-rts","title":"Neural Fictitious Self-Play on ELF Mini-RTS","date":"2019-02-06","arxiv_id":"1902.02004","repositories_listed":0,"syntology":null},{"url":null,"slug":"space-navigator-a-tool-for-the-optimization","title":"Space Navigator: a Tool for the Optimization of Collision Avoidance Maneuvers","date":"2019-02-06","arxiv_id":"1902.02095","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-for-autonomous","title":"Adaptive Stress Testing for Autonomous Vehicles","date":"2019-02-05","arxiv_id":"1902.01909","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphastar-an-evolutionary-computation","title":"AlphaStar: An Evolutionary Computation Perspective","date":"2019-02-05","arxiv_id":"1902.01724","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactively-shaping-robot-behaviour-with","title":"Interactively shaping robot behaviour with unlabeled human instructions","date":"2019-02-05","arxiv_id":"1902.01670","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-learn-in-simulation","title":"Learning to Learn in Simulation","date":"2019-02-05","arxiv_id":"1902.01569","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyphonic-music-composition-with-lstm-neural","title":"Polyphonic Music Composition with LSTM Neural Networks and Reinforcement Learning","date":"2019-02-05","arxiv_id":"1902.01973","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimal-load","title":"Reinforcement Learning for Optimal Load Distribution Sequencing in Resource-Sharing System","date":"2019-02-05","arxiv_id":"1902.01899","repositories_listed":0,"syntology":null},{"url":null,"slug":"total-stochastic-gradient-algorithms-and","title":"Total stochastic gradient algorithms and applications in reinforcement learning","date":"2019-02-05","arxiv_id":"1902.01722","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-multimodal-multitask-learning","title":"Embodied Multimodal Multitask Learning","date":"2019-02-04","arxiv_id":"1902.01385","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-aware-recommendation-based-on","title":"Value-aware Recommendation based on Reinforced Profit Maximization in E-commerce Systems","date":"2019-02-03","arxiv_id":"1902.00851","repositories_listed":0,"syntology":null},{"url":null,"slug":"belief-dynamics-extraction","title":"Belief dynamics extraction","date":"2019-02-02","arxiv_id":"1902.00673","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-user-preferences-via-reinforcement","title":"Learning User Preferences via Reinforcement Learning with Spatial Interface Valuing","date":"2019-02-02","arxiv_id":"1902.00719","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-asymptotic-analysis-of-biased-stochastic","title":"Non-asymptotic Analysis of Biased Stochastic Approximation Scheme","date":"2019-02-02","arxiv_id":"1902.00629","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-collaborative-filtering-meets","title":"When Collaborative Filtering Meets Reinforcement Learning","date":"2019-02-02","arxiv_id":"1902.00715","repositories_listed":0,"syntology":null},{"url":null,"slug":"competitive-experience-replay","title":"Competitive Experience Replay","date":"2019-02-01","arxiv_id":"1902.00528","repositories_listed":0,"syntology":null},{"url":"/paper/joint-entity-linking-with-deep-reinforcement","slug":"joint-entity-linking-with-deep-reinforcement","title":"Joint Entity Linking with Deep Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00330","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-action-representations-for","title":"Learning Action Representations for Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00183","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-off-policy-evaluation","title":"Privacy Preserving Off-Policy Evaluation","date":"2019-02-01","arxiv_id":"1902.00174","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-rationalizations-in-deep-reinforcement","title":"Visual Rationalizations in Deep Reinforcement Learning for Atari Games","date":"2019-02-01","arxiv_id":"1902.00566","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-geometric-perspective-on-optimal","title":"A Geometric Perspective on Optimal Representations for Reinforcement Learning","date":"2019-01-31","arxiv_id":"1901.11530","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theory-of-regularized-markov-decision","title":"A Theory of Regularized Markov Decision Processes","date":"2019-01-31","arxiv_id":"1901.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"accuracy-vs-efficiency-achieving-both-through","title":"Accuracy vs. Efficiency: Achieving Both through FPGA-Implementation Aware Neural Architecture Search","date":"2019-01-31","arxiv_id":"1901.11211","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimization-framework-for-task-sequencing","title":"An Optimization Framework for Task Sequencing in Curriculum Learning","date":"2019-01-31","arxiv_id":"1901.11478","repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-features-support-model-based-and","title":"Successor Features Combine Elements of Model-Free and Model-based Reinforcement Learning","date":"2019-01-31","arxiv_id":"1901.11437","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-value-function-polytope-in-reinforcement","title":"The Value Function Polytope in Reinforcement Learning","date":"2019-01-31","arxiv_id":"1901.11524","repositories_listed":0,"syntology":null},{"url":null,"slug":"tsallis-reinforcement-learning-a-unified","title":"Tsallis Reinforcement Learning: A Unified Framework for Maximum Entropy Reinforcement Learning","date":"2019-01-31","arxiv_id":"1902.00137","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-expected-and","title":"A Comparative Analysis of Expected and Distributional Reinforcement Learning","date":"2019-01-30","arxiv_id":"1901.11084","repositories_listed":0,"syntology":null},{"url":null,"slug":"infobot-transfer-and-exploration-via-the","title":"InfoBot: Transfer and Exploration via the Information Bottleneck","date":"2019-01-30","arxiv_id":"1901.10902","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-position-evaluation-functions-used","title":"Learning Position Evaluation Functions Used in Monte Carlo Softmax Search","date":"2019-01-30","arxiv_id":"1901.10706","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-pandoras-boxes-and-bandits","title":"Online Pandora's Boxes and Bandits","date":"2019-01-30","arxiv_id":"1901.10698","repositories_listed":0,"syntology":null},{"url":null,"slug":"probability-functional-descent-a-unifying","title":"Probability Functional Descent: A Unifying Perspective on GANs, Variational Inference, and Reinforcement Learning","date":"2019-01-30","arxiv_id":"1901.10691","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-in-deep-reinforcement-learning-using","title":"Transfer in Deep Reinforcement Learning Using Successor Features and Generalised Policy Improvement","date":"2019-01-30","arxiv_id":"1901.10964","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-regulation-enforcement-solution-for-multi","title":"A Regulation Enforcement Solution for Multi-agent Reinforcement Learning","date":"2019-01-29","arxiv_id":"1901.10059","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-multi","title":"Multi-Agent Reinforcement Learning with Multi-Step Generative Models","date":"2019-01-29","arxiv_id":"1901.10251","repositories_listed":0,"syntology":null},{"url":null,"slug":"clic-curriculum-learning-and-imitation-for","title":"CLIC: Curriculum Learning and Imitation for object Control in non-rewarding environments","date":"2019-01-28","arxiv_id":"1901.09720","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-a-multi-objective-reward-function","title":"Designing a Multi-Objective Reward Function for Creating Teams of Robotic Bodyguards Using Deep Reinforcement Learning","date":"2019-01-28","arxiv_id":"1901.09837","repositories_listed":0,"syntology":null},{"url":null,"slug":"modularization-of-end-to-end-learning-case","title":"Modularization of End-to-End Learning: Case Study in Arcade Games","date":"2019-01-27","arxiv_id":"1901.09895","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-deep-reinforcement-learning-by","title":"Off-Policy Deep Reinforcement Learning by Bootstrapping the Covariate Shift","date":"2019-01-27","arxiv_id":"1901.09455","repositories_listed":0,"syntology":null}],"record_sha256":"8193f0bc2dfcb3380514a1584cd838cb784bceb6fdeab4e9385b74aaf4206688","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}