{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/137","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":137,"pages_in_order":152,"rows_per_page":100,"rows":[13601,13700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/136","next":"/task/reinforcement-learning-1/papers/138","papers":[{"url":null,"slug":"reinforcement-learning-based-curriculum","title":"Reinforcement Learning based Curriculum Optimization for Neural Machine Translation","date":"2019-02-28","arxiv_id":"1903.00041","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-caching-via-deep-reinforcement","title":"Deep Reinforcement Learning for Adaptive Caching in Hierarchical Content Delivery Networks","date":"2019-02-27","arxiv_id":"1902.10301","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-edge-caching-via-reinforcement","title":"Distributed Edge Caching via Reinforcement Learning in Fog Radio Access Networks","date":"2019-02-27","arxiv_id":"1902.10574","repositories_listed":0,"syntology":null},{"url":null,"slug":"introspection-learning","title":"Introspection Learning","date":"2019-02-27","arxiv_id":"1902.10754","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-packet-classification","title":"Neural Packet Classification","date":"2019-02-27","arxiv_id":"1902.10319","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-ensemble-methods-for-q-learning-via","title":"Unifying Ensemble Methods for Q-learning via Social Choice Theory","date":"2019-02-27","arxiv_id":"1902.10646","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-meta-interpretive-learning-outperform","title":"Can Meta-Interpretive Learning outperform Deep Reinforcement Learning of Evaluable Game strategies?","date":"2019-02-26","arxiv_id":"1902.09835","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-in-hierarchical-reinforcement","title":"Planning in Hierarchical Reinforcement Learning: Guarantees for Using Local Policies","date":"2019-02-26","arxiv_id":"1902.10140","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-agent-incentives-using-causal","title":"Understanding Agent Incentives using Causal Influence Diagrams. Part I: Single Action Settings","date":"2019-02-26","arxiv_id":"1902.09980","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-ternary-rewards-to-reason-over","title":"Learning When Not to Answer: A Ternary Reward Structure for Reinforcement Learning based Question Answering","date":"2019-02-26","arxiv_id":"1902.10236","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-under","title":"Adversarial Reinforcement Learning under Partial Observability in Autonomous Computer Network Defence","date":"2019-02-25","arxiv_id":"1902.09062","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-extreme-hummingbird-maneuvers-on","title":"Learning Extreme Hummingbird Maneuvers on Flapping Wing Robots","date":"2019-02-25","arxiv_id":"1902.09626","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-range-indoor-navigation-with-prm-rl","title":"Long-Range Indoor Navigation with PRM-RL","date":"2019-02-25","arxiv_id":"1902.09458","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-trigger-continual-state-representation","title":"S-TRIGGER: Continual State Representation Learning via Self-Triggered Generative Replay","date":"2019-02-25","arxiv_id":"1902.09434","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggregating-e-commerce-search-results-from","title":"Aggregating E-commerce Search Results from Heterogeneous Sources via Hierarchical Reinforcement Learning","date":"2019-02-24","arxiv_id":"1902.08882","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-reinforcement","title":"Distributionally Robust Reinforcement Learning","date":"2019-02-23","arxiv_id":"1902.08708","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-memory-for-lifelong-reinforcement","title":"Generative Memory for Lifelong Reinforcement Learning","date":"2019-02-22","arxiv_id":"1902.08349","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-deterministic-policy-with-target-for","title":"Learning Deterministic Policy with Target for Power Control in Wireless Networks","date":"2019-02-21","arxiv_id":"1902.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistics-and-samples-in-distributional","title":"Statistics and Samples in Distributional Reinforcement Learning","date":"2019-02-21","arxiv_id":"1902.08102","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-experience-prioritization","title":"Curiosity-Driven Experience Prioritization via Density Estimation","date":"2019-02-20","arxiv_id":"1902.08039","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-language-to-goals-inverse-reinforcement","title":"From Language to Goals: Inverse Reinforcement Learning for Vision-Based Instruction Following","date":"2019-02-20","arxiv_id":"1902.07742","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-repetition-normalized-adversarial","title":"A novel repetition normalized adversarial reward for headline generation","date":"2019-02-19","arxiv_id":"1902.07110","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-coordination-through-competition","title":"Emergent Coordination Through Competition","date":"2019-02-19","arxiv_id":"1902.07151","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-generalisation-in-continuous","title":"Investigating Generalisation in Continuous Deep Reinforcement Learning","date":"2019-02-19","arxiv_id":"1902.07015","repositories_listed":0,"syntology":null},{"url":null,"slug":"message-dropout-an-efficient-training-method","title":"Message-Dropout: An Efficient Training Method for Multi-Agent Deep Reinforcement Learning","date":"2019-02-18","arxiv_id":"1902.06527","repositories_listed":0,"syntology":null},{"url":null,"slug":"parenting-safe-reinforcement-learning-from","title":"Parenting: Safe Reinforcement Learning from Human Input","date":"2019-02-18","arxiv_id":"1902.06766","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-potential-based-reward-shaping-for","title":"A new Potential-Based Reward Shaping for Reinforcement Learning Agent","date":"2019-02-17","arxiv_id":"1902.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-topologies-between-learning","title":"Leveraging Communication Topologies Between Learning Agents in Deep Reinforcement Learning","date":"2019-02-16","arxiv_id":"1902.06740","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-coagent-networks-stochastic","title":"Asynchronous Coagent Networks","date":"2019-02-15","arxiv_id":"1902.05650","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-high-level","title":"Deep Reinforcement Learning Based High-level Driving Behavior Decision-making Model in Heterogeneous Traffic","date":"2019-02-15","arxiv_id":"1902.05772","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-in-pomdps-with","title":"Robust Reinforcement Learning in POMDPs with Incomplete and Noisy Observations","date":"2019-02-15","arxiv_id":"1902.05795","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-perception-in-adversarial-scenarios","title":"Active Perception in Adversarial Scenarios using Maximum Entropy Deep Reinforcement Learning","date":"2019-02-14","arxiv_id":"1902.05644","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-ua-v-attitude","title":"Reinforcement Learning for UA V Attitude Control","date":"2019-02-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-to-optimize-long-term","title":"Reinforcement Learning to Optimize Long-term User Engagement in Recommender Systems","date":"2019-02-13","arxiv_id":"1902.05570","repositories_listed":0,"syntology":null},{"url":null,"slug":"actrce-augmenting-experience-via-teachers","title":"ACTRCE: Augmenting Experience via Teacher's Advice For Multi-Goal Reinforcement Learning","date":"2019-02-12","arxiv_id":"1902.04546","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-from-policy","title":"Deep Reinforcement Learning from Policy-Dependent Human Feedback","date":"2019-02-12","arxiv_id":"1902.04257","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-space-reinforcement-learning-for","title":"Latent Space Reinforcement Learning for Steering Angle Prediction","date":"2019-02-11","arxiv_id":"1902.03765","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-whole","title":"Whole-Chain Recommendations","date":"2019-02-11","arxiv_id":"1902.03987","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-dynamics-and-termination-errors","title":"Performance Dynamics and Termination Errors in Reinforcement Learning: A Unifying Perspective","date":"2019-02-11","arxiv_id":"1902.04179","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-reinforcement-learning","title":"Stochastic Reinforcement Learning","date":"2019-02-11","arxiv_id":"1902.04178","repositories_listed":0,"syntology":null},{"url":null,"slug":"wisemove-a-framework-for-safe-deep","title":"WiseMove: A Framework for Safe Deep Reinforcement Learning for Autonomous Driving","date":"2019-02-11","arxiv_id":"1902.04118","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bandit-framework-for-optimal-selection-of","title":"A Bandit Framework for Optimal Selection of Reinforcement Learning Agents","date":"2019-02-10","arxiv_id":"1902.03657","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with","title":"Distributional reinforcement learning with linear function approximation","date":"2019-02-08","arxiv_id":"1902.03149","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-discount-factor-in","title":"Rethinking the Discount Factor in Reinforcement Learning: A Decision Theoretic Approach","date":"2019-02-08","arxiv_id":"1902.02893","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaoptimization-on-a-distributed-system-for","title":"Metaoptimization on a Distributed System for Deep Reinforcement Learning","date":"2019-02-07","arxiv_id":"1902.02725","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-search-and-recognition-for-robot-task","title":"Visual search and recognition for robot task execution and monitoring","date":"2019-02-07","arxiv_id":"1902.02870","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-guiding-principle-for-causal-decision","title":"A Guiding Principle for Causal Decision Problems","date":"2019-02-06","arxiv_id":"1902.02279","repositories_listed":0,"syntology":null},{"url":null,"slug":"cesma-centralized-expert-supervises-multi","title":"Decentralized Multi-Agents by Imitation of a Centralized Controller","date":"2019-02-06","arxiv_id":"1902.02311","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-fictitious-self-play-on-elf-mini-rts","title":"Neural Fictitious Self-Play on ELF Mini-RTS","date":"2019-02-06","arxiv_id":"1902.02004","repositories_listed":0,"syntology":null},{"url":null,"slug":"space-navigator-a-tool-for-the-optimization","title":"Space Navigator: a Tool for the Optimization of Collision Avoidance Maneuvers","date":"2019-02-06","arxiv_id":"1902.02095","repositories_listed":0,"syntology":null},{"url":null,"slug":"weak-consistency-of-the-1-nearest-neighbor","title":"On $L_2$-consistency of nearest neighbor matching","date":"2019-02-06","arxiv_id":"1902.02408","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-for-autonomous","title":"Adaptive Stress Testing for Autonomous Vehicles","date":"2019-02-05","arxiv_id":"1902.01909","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphastar-an-evolutionary-computation","title":"AlphaStar: An Evolutionary Computation Perspective","date":"2019-02-05","arxiv_id":"1902.01724","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactively-shaping-robot-behaviour-with","title":"Interactively shaping robot behaviour with unlabeled human instructions","date":"2019-02-05","arxiv_id":"1902.01670","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyphonic-music-composition-with-lstm-neural","title":"Polyphonic Music Composition with LSTM Neural Networks and Reinforcement Learning","date":"2019-02-05","arxiv_id":"1902.01973","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimal-load","title":"Reinforcement Learning for Optimal Load Distribution Sequencing in Resource-Sharing System","date":"2019-02-05","arxiv_id":"1902.01899","repositories_listed":0,"syntology":null},{"url":null,"slug":"total-stochastic-gradient-algorithms-and","title":"Total stochastic gradient algorithms and applications in reinforcement learning","date":"2019-02-05","arxiv_id":"1902.01722","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-aware-recommendation-based-on","title":"Value-aware Recommendation based on Reinforced Profit Maximization in E-commerce Systems","date":"2019-02-03","arxiv_id":"1902.00851","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-user-preferences-via-reinforcement","title":"Learning User Preferences via Reinforcement Learning with Spatial Interface Valuing","date":"2019-02-02","arxiv_id":"1902.00719","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-asymptotic-analysis-of-biased-stochastic","title":"Non-asymptotic Analysis of Biased Stochastic Approximation Scheme","date":"2019-02-02","arxiv_id":"1902.00629","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-collaborative-filtering-meets","title":"When Collaborative Filtering Meets Reinforcement Learning","date":"2019-02-02","arxiv_id":"1902.00715","repositories_listed":0,"syntology":null},{"url":null,"slug":"competitive-experience-replay","title":"Competitive Experience Replay","date":"2019-02-01","arxiv_id":"1902.00528","repositories_listed":0,"syntology":null},{"url":"/paper/joint-entity-linking-with-deep-reinforcement","slug":"joint-entity-linking-with-deep-reinforcement","title":"Joint Entity Linking with Deep Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00330","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-action-representations-for","title":"Learning Action Representations for Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00183","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-off-policy-evaluation","title":"Privacy Preserving Off-Policy Evaluation","date":"2019-02-01","arxiv_id":"1902.00174","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-rationalizations-in-deep-reinforcement","title":"Visual Rationalizations in Deep Reinforcement Learning for Atari Games","date":"2019-02-01","arxiv_id":"1902.00566","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-geometric-perspective-on-optimal","title":"A Geometric Perspective on Optimal Representations for Reinforcement Learning","date":"2019-01-31","arxiv_id":"1901.11530","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimization-framework-for-task-sequencing","title":"An Optimization Framework for Task Sequencing in Curriculum Learning","date":"2019-01-31","arxiv_id":"1901.11478","repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-features-support-model-based-and","title":"Successor Features Combine Elements of Model-Free and Model-based Reinforcement Learning","date":"2019-01-31","arxiv_id":"1901.11437","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-value-function-polytope-in-reinforcement","title":"The Value Function Polytope in Reinforcement Learning","date":"2019-01-31","arxiv_id":"1901.11524","repositories_listed":0,"syntology":null},{"url":null,"slug":"tsallis-reinforcement-learning-a-unified","title":"Tsallis Reinforcement Learning: A Unified Framework for Maximum Entropy Reinforcement Learning","date":"2019-01-31","arxiv_id":"1902.00137","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-expected-and","title":"A Comparative Analysis of Expected and Distributional Reinforcement Learning","date":"2019-01-30","arxiv_id":"1901.11084","repositories_listed":0,"syntology":null},{"url":null,"slug":"probability-functional-descent-a-unifying","title":"Probability Functional Descent: A Unifying Perspective on GANs, Variational Inference, and Reinforcement Learning","date":"2019-01-30","arxiv_id":"1901.10691","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-in-deep-reinforcement-learning-using","title":"Transfer in Deep Reinforcement Learning Using Successor Features and Generalised Policy Improvement","date":"2019-01-30","arxiv_id":"1901.10964","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-regulation-enforcement-solution-for-multi","title":"A Regulation Enforcement Solution for Multi-agent Reinforcement Learning","date":"2019-01-29","arxiv_id":"1901.10059","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-multi","title":"Multi-Agent Reinforcement Learning with Multi-Step Generative Models","date":"2019-01-29","arxiv_id":"1901.10251","repositories_listed":0,"syntology":null},{"url":null,"slug":"clic-curriculum-learning-and-imitation-for","title":"CLIC: Curriculum Learning and Imitation for object Control in non-rewarding environments","date":"2019-01-28","arxiv_id":"1901.09720","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-a-multi-objective-reward-function","title":"Designing a Multi-Objective Reward Function for Creating Teams of Robotic Bodyguards Using Deep Reinforcement Learning","date":"2019-01-28","arxiv_id":"1901.09837","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-deep-reinforcement-learning-by","title":"Off-Policy Deep Reinforcement Learning by Bootstrapping the Covariate Shift","date":"2019-01-27","arxiv_id":"1901.09455","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shaping-via-meta-learning","title":"Reward Shaping via Meta-Learning","date":"2019-01-27","arxiv_id":"1901.09330","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-propagation-for-decentralized-networked","title":"Value Propagation for Decentralized Networked Deep Multi-agent Reinforcement Learning","date":"2019-01-27","arxiv_id":"1901.09326","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-recursive-reasoning-for-multi","title":"Probabilistic Recursive Reasoning for Multi-Agent Reinforcement Learning","date":"2019-01-26","arxiv_id":"1901.09207","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-deep-reinforcement-learning-for","title":"Model-based Deep Reinforcement Learning for Dynamic Portfolio Optimization","date":"2019-01-25","arxiv_id":"1901.08740","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning","title":"Federated Deep Reinforcement Learning","date":"2019-01-24","arxiv_id":"1901.08277","repositories_listed":0,"syntology":null},{"url":null,"slug":"feudal-multi-agent-hierarchies-for","title":"Feudal Multi-Agent Hierarchies for Cooperative Reinforcement Learning","date":"2019-01-24","arxiv_id":"1901.08492","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-estimating-the-policy","title":"Sample Complexity of Estimating the Policy Gradient for Nearly Deterministic Dynamical Systems","date":"2019-01-24","arxiv_id":"1901.08562","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillation-strategies-for-proximal-policy","title":"Distillation Strategies for Proximal Policy Optimization","date":"2019-01-23","arxiv_id":"1901.08128","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-multi","title":"Hierarchical Reinforcement Learning for Multi-agent MOBA Game","date":"2019-01-23","arxiv_id":"1901.08004","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-enriched-text-representation-for","title":"Phonetic-enriched Text Representation for Chinese Sentiment Analysis with Reinforcement Learning","date":"2019-01-23","arxiv_id":"1901.07880","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-markov-decision","title":"Reinforcement Learning of Markov Decision Processes with Peak Constraints","date":"2019-01-23","arxiv_id":"1901.07839","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-recovery-controller-for-a-quadrupedal","title":"Robust Recovery Controller for a Quadrupedal Robot using Deep Reinforcement Learning","date":"2019-01-22","arxiv_id":"1901.07517","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-imitation-learning-with-recurrent","title":"Towards Learning to Imitate from a Single Video Demonstration","date":"2019-01-22","arxiv_id":"1901.07186","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-survey-on-probabilistic-reinforcement","title":"A Short Survey on Probabilistic Reinforcement Learning","date":"2019-01-21","arxiv_id":"1901.07010","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-federated-reinforcement-learning-a","title":"Lifelong Federated Reinforcement Learning: A Learning Architecture for Navigation in Cloud Robotic Systems","date":"2019-01-19","arxiv_id":"1901.06455","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-physically-safe-reinforcement","title":"Towards Physically Safe Reinforcement Learning under Supervision","date":"2019-01-19","arxiv_id":"1901.06576","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-embedded","title":"Multi-agent Reinforcement Learning Embedded Game for the Optimization of Building Energy Control and Power System Planning","date":"2019-01-17","arxiv_id":"1901.07333","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionarily-curated-curriculum-learning","title":"Evolutionarily-Curated Curriculum Learning for Deep Reinforcement Learning Agents","date":"2019-01-16","arxiv_id":"1901.05431","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-on-graphs-a","title":"Representation Learning on Graphs: A Reinforcement Learning Application","date":"2019-01-16","arxiv_id":"1901.05351","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-knowledge-based-reinforcement","title":"Comparing Knowledge-based Reinforcement Learning to Neural Networks in a Strategy Game","date":"2019-01-15","arxiv_id":"1901.04626","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-sepsis-treatment-strategies-by","title":"Improving Sepsis Treatment Strategies by Combining Deep and Kernel-Based Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04670","repositories_listed":0,"syntology":null}],"record_sha256":"cb6b4aa9a744199c36bac59eb02b33362e3f0d3dcd3f5302e84001064fb3bfb3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}