{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/118","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":118,"pages_in_order":135,"rows_per_page":100,"rows":[11701,11800],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/117","next":"/task/reinforcement-learning-2/papers/119","papers":[{"url":null,"slug":"skill-transfer-in-deep-reinforcement-learning","title":"Skill Transfer in Deep Reinforcement Learning under Morphological Heterogeneity","date":"2019-08-14","arxiv_id":"1908.05265","repositories_listed":0,"syntology":null},{"url":null,"slug":"competitive-multi-agent-deep-reinforcement","title":"Competitive Multi-Agent Deep Reinforcement Learning with Counterfactual Thinking","date":"2019-08-13","arxiv_id":"1908.04573","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-crystallized-adaptivity-to-fluid","title":"From Crystallized Adaptivity to Fluid Adaptivity in Deep Reinforcement Learning -- Insights from Biological Systems on Adaptive Flexibility","date":"2019-08-13","arxiv_id":"1908.05348","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-tampering-problems-and-solutions-in","title":"Reward Tampering Problems and Solutions in Reinforcement Learning: A Causal Influence Diagram Perspective","date":"2019-08-13","arxiv_id":"1908.04734","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-adaptation-with-meta-reinforcement","title":"Fast Adaptation with Meta-Reinforcement Learning for Trust Modelling in Human-Robot Interaction","date":"2019-08-12","arxiv_id":"1908.04087","repositories_listed":0,"syntology":null},{"url":null,"slug":"superstition-in-the-network-deep","title":"Superstition in the Network: Deep Reinforcement Learning Plays Deceptive Games","date":"2019-08-12","arxiv_id":"1908.04436","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-cooperative-multi-agent-deep","title":"A Review of Cooperative Multi-Agent Deep Reinforcement Learning","date":"2019-08-11","arxiv_id":"1908.03963","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-traffic-signal-control-using-a","title":"Large-Scale Traffic Signal Control Using a Novel Multi-Agent Reinforcement Learning","date":"2019-08-10","arxiv_id":"1908.03761","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-reinforcement-learning-a-new","title":"Incremental Reinforcement Learning --- a New Continuous Reinforcement Learning Frame Based on Stochastic Differential Equation methods","date":"2019-08-08","arxiv_id":"1908.02974","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-grasp-from-25d-images-a-deep","title":"Learning to Grasp from 2.5D images: a Deep Reinforcement Learning Approach","date":"2019-08-08","arxiv_id":"1908.03440","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-stochastic-game-theory-approach-for-the","title":"A physics-informed reinforcement learning approach for the interfacial area transport in two-phase flow","date":"2019-08-06","arxiv_id":"1908.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-recurrent-q-learning-for-backchannel","title":"Batch Recurrent Q-Learning for Backchannel Generation Towards Engaging Agents","date":"2019-08-06","arxiv_id":"1908.02037","repositories_listed":0,"syntology":null},{"url":null,"slug":"promoting-coordination-through-policy","title":"Promoting Coordination through Policy Regularization in Multi-Agent Deep Reinforcement Learning","date":"2019-08-06","arxiv_id":"1908.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"construction-of-macro-actions-for-deep","title":"Reusability and Transferability of Macro Actions for Reinforcement Learning","date":"2019-08-05","arxiv_id":"1908.01478","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-driven-backchannel-generation-using","title":"Speech Driven Backchannel Generation using Deep Q-Network for Enhancing Engagement in Human-Robot Interaction","date":"2019-08-05","arxiv_id":"1908.01618","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-system","title":"A View on Deep Reinforcement Learning in System Optimization","date":"2019-08-04","arxiv_id":"1908.01275","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-deep-reinforcement-learning-in","title":"Improving Deep Reinforcement Learning in Minecraft with Action Advice","date":"2019-08-02","arxiv_id":"1908.01007","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-reinforcement-learning-for","title":"Curiosity-driven Reinforcement Learning for Diverse Visual Paragraph Generation","date":"2019-08-01","arxiv_id":"1908.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-when-to-drive-in-intersections-by","title":"Learning When to Drive in Intersections by Combining Reinforcement Learning and Model Predictive Control","date":"2019-08-01","arxiv_id":"1908.00177","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-personalized","title":"Reinforcement Learning for Personalized Dialogue Management","date":"2019-08-01","arxiv_id":"1908.00286","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-multiple","title":"Inverse Reinforcement Learning with Multiple Ranked Experts","date":"2019-07-31","arxiv_id":"1907.13411","repositories_listed":0,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-based","slug":"multi-agent-reinforcement-learning-based","title":"Multi-Agent Reinforcement Learning Based Frame Sampling for Effective Untrimmed Video Recognition","date":"2019-07-31","arxiv_id":"1907.13369","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-attacks-on-reinforcement-learning","title":"Optimal Attacks on Reinforcement Learning Policies","date":"2019-07-31","arxiv_id":"1907.13548","repositories_listed":0,"syntology":null},{"url":null,"slug":"precodernet-hybrid-beamforming-for-millimeter","title":"PrecoderNet: Hybrid Beamforming for Millimeter Wave Systems with Deep Reinforcement Learning","date":"2019-07-31","arxiv_id":"1907.13266","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepplace-learning-to-place-applications-in","title":"DeepPlace: Learning to Place Applications in Multi-Tenant Clusters","date":"2019-07-30","arxiv_id":"1907.12916","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-unsupervised-learning-for","title":"Model-Free Unsupervised Learning for Optimization Problems with Constraints","date":"2019-07-30","arxiv_id":"1907.12706","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-robust-reinforcement-learning","title":"Wasserstein Robust Reinforcement Learning","date":"2019-07-30","arxiv_id":"1907.13196","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-driven-sequential-data-abstraction","title":"Goal-Driven Sequential Data Abstraction","date":"2019-07-29","arxiv_id":"1907.12336","repositories_listed":0,"syntology":null},{"url":null,"slug":"taxable-stock-trading-with-deep-reinforcement","title":"Taxable Stock Trading with Deep Reinforcement Learning","date":"2019-07-28","arxiv_id":"1907.12093","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-personalized","title":"Deep Reinforcement Learning for Personalized Search Story Recommendation","date":"2019-07-26","arxiv_id":"1907.11754","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-continuous-time-mean-variance","title":"Large scale continuous-time mean-variance portfolio allocation via reinforcement learning","date":"2019-07-26","arxiv_id":"1907.11718","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-hard-exploration-for-reinforcement","title":"On Hard Exploration for Reinforcement Learning: a Case Study in Pommerman","date":"2019-07-26","arxiv_id":"1907.11788","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-guidance-with-mcts-for-deep","title":"Action Guidance with MCTS for Deep Reinforcement Learning","date":"2019-07-25","arxiv_id":"1907.11703","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-input-for-deep-reinforcement-learning","title":"Dynamic Input for Deep Reinforcement Learning in Autonomous Driving","date":"2019-07-25","arxiv_id":"1907.10994","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-lungs-auscultation-with","title":"Interactive Lungs Auscultation with Reinforcement Learning Agent","date":"2019-07-25","arxiv_id":"1907.11238","repositories_listed":0,"syntology":null},{"url":null,"slug":"alphastock-a-buying-winners-and-selling","title":"AlphaStock: A Buying-Winners-and-Selling-Losers Investment Strategy using Interpretable Deep Reinforcement Attention Networks","date":"2019-07-24","arxiv_id":"1908.02646","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-reinforcement-learning-2","title":"Fairness in Reinforcement Learning","date":"2019-07-24","arxiv_id":"1907.10323","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-goal-oriented-visual-dialog-agents","title":"Learning Goal-Oriented Visual Dialog Agents: Imitating and Surpassing Analytic Experts","date":"2019-07-24","arxiv_id":"1907.10500","repositories_listed":0,"syntology":null},{"url":null,"slug":"terminal-prediction-as-an-auxiliary-task-for","title":"Terminal Prediction as an Auxiliary Task for Deep Reinforcement Learning","date":"2019-07-24","arxiv_id":"1907.10827","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-modeling-as-auxiliary-task-for-deep","title":"Agent Modeling as Auxiliary Task for Deep Reinforcement Learning","date":"2019-07-22","arxiv_id":"1907.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-autonomous-1","title":"Deep Reinforcement Learning for Autonomous Internet of Things: Model, Applications and Challenges","date":"2019-07-22","arxiv_id":"1907.09059","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-clinical","title":"Deep Reinforcement Learning for Clinical Decision Support: A Brief Survey","date":"2019-07-22","arxiv_id":"1907.09475","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-policy-learning-for-non-stationary","title":"Efficient Policy Learning for Non-Stationary MDPs under Adversarial Manipulation","date":"2019-07-22","arxiv_id":"1907.09350","repositories_listed":0,"syntology":null},{"url":null,"slug":"surrogate-models-for-enhancing-the-efficiency","title":"Surrogate Models for Enhancing the Efficiency of Neuroevolution in Reinforcement Learning","date":"2019-07-22","arxiv_id":"1907.09300","repositories_listed":0,"syntology":null},{"url":null,"slug":"vrls-a-unified-reinforcement-learning","title":"VRLS: A Unified Reinforcement Learning Scheduler for Vehicle-to-Vehicle Communications","date":"2019-07-22","arxiv_id":"1907.09319","repositories_listed":0,"syntology":null},{"url":null,"slug":"techniques-for-automated-machine-learning","title":"Techniques for Automated Machine Learning","date":"2019-07-21","arxiv_id":"1907.08908","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-actor-critic-attention-mechanism-for-deep","title":"An Actor-Critic-Attention Mechanism for Deep Reinforcement Learning in Multi-view Environments","date":"2019-07-19","arxiv_id":"1907.09466","repositories_listed":0,"syntology":null},{"url":null,"slug":"delegative-reinforcement-learning-learning-to","title":"Delegative Reinforcement Learning: learning to avoid traps with a little help","date":"2019-07-19","arxiv_id":"1907.08461","repositories_listed":0,"syntology":null},{"url":null,"slug":"combinatorial-keyword-recommendations-for","title":"Combinatorial Keyword Recommendations for Sponsored Search with Deep Reinforcement Learning","date":"2019-07-18","arxiv_id":"1907.08686","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamical-distance-learning-for-unsupervised","title":"Dynamical Distance Learning for Semi-Supervised and Unsupervised Skill Discovery","date":"2019-07-18","arxiv_id":"1907.08225","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritized-guidance-for-efficient-multi","title":"Prioritized Guidance for Efficient Multi-Agent Reinforcement Learning Exploration","date":"2019-07-18","arxiv_id":"1907.07847","repositories_listed":0,"syntology":null},{"url":null,"slug":"photonic-architecture-for-reinforcement","title":"Photonic architecture for reinforcement learning","date":"2019-07-17","arxiv_id":"1907.07503","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-inductive-synthesis-framework-for","title":"An Inductive Synthesis Framework for Verifiable Reinforcement Learning","date":"2019-07-16","arxiv_id":"1907.07273","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-reinforcement-learning-through","title":"Improved Reinforcement Learning through Imitation Learning Pretraining Towards Image-based Autonomous Driving","date":"2019-07-16","arxiv_id":"1907.06838","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-control-of-chaos-with-continuous","title":"Model-free Control of Chaos with Continuous Deep Q-learning","date":"2019-07-16","arxiv_id":"1907.07775","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-memory-structure-for-efficient-use-of","title":"A Dual Memory Structure for Efficient Use of Replay Memory in Deep Reinforcement Learning","date":"2019-07-15","arxiv_id":"1907.06396","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-distillation-with","title":"Federated Reinforcement Distillation with Proxy Experience Memory","date":"2019-07-15","arxiv_id":"1907.06536","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutual-reinforcement-learning","title":"Mutual Reinforcement Learning","date":"2019-07-15","arxiv_id":"1907.06725","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranking-sentences-from-product-description","title":"Ranking sentences from product description & bullets for better search","date":"2019-07-15","arxiv_id":"1907.06330","repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-reconstruction-with-hidden","title":"Environment Reconstruction with Hidden Confounders for Reinforcement Learning based Recommendation","date":"2019-07-12","arxiv_id":"1907.06584","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-model-based-approach-for-sample-efficient","title":"A Model-based Approach for Sample-efficient Multi-task Reinforcement Learning","date":"2019-07-11","arxiv_id":"1907.04964","repositories_listed":0,"syntology":null},{"url":null,"slug":"discorl-continual-reinforcement-learning-via","title":"DisCoRL: Continual Reinforcement Learning via Policy Distillation","date":"2019-07-11","arxiv_id":"1907.05855","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-projected-policy-gradient-for","title":"Imitation-Projected Programmatic Reinforcement Learning","date":"2019-07-11","arxiv_id":"1907.05431","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-based-driving","title":"Deep Reinforcement-Learning-based Driving Policy for Autonomous Road Vehicles","date":"2019-07-10","arxiv_id":"1907.05246","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-transferability-from-simulation-to","title":"Assessing Transferability from Simulation to Reality for Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04685","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-dynamics-models-for-data","title":"Interpretable Dynamics Models for Data-Efficient Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04902","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-chromatic","title":"Reinforcement Learning with Chromatic Networks for Compact Architecture Search","date":"2019-07-10","arxiv_id":"1907.06511","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-financial","title":"Capturing Financial markets to apply Deep Reinforcement Learning","date":"2019-07-09","arxiv_id":"1907.04373","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreaming-machine-learning-lipschitz","title":"Dreaming machine learning: Lipschitz extensions for reinforcement learning on financial markets","date":"2019-07-09","arxiv_id":"1907.05697","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-reinforcement-learning-for","title":"Data Efficient Reinforcement Learning for Legged Robots","date":"2019-07-08","arxiv_id":"1907.03613","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-algorithms-have-no-guarantees","title":"Policy-Gradient Algorithms Have No Guarantees of Convergence in Linear Quadratic Games","date":"2019-07-08","arxiv_id":"1907.03712","repositories_listed":0,"syntology":null},{"url":null,"slug":"shrinkml-end-to-end-asr-model-compression","title":"ShrinkML: End-to-End ASR Model Compression Using Reinforcement Learning","date":"2019-07-08","arxiv_id":"1907.03540","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-inference-mpc-for-bayesian-model","title":"Variational Inference MPC for Bayesian Model-based Reinforcement Learning","date":"2019-07-08","arxiv_id":"1907.04202","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-communication-efficient-multi-agent-actor","title":"A Communication-Efficient Multi-Agent Actor-Critic Algorithm for Distributed Reinforcement Learning","date":"2019-07-06","arxiv_id":"1907.03053","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsic-motivation-driven-intuitive-physics","title":"Intrinsic Motivation Driven Intuitive Physics Learning using Deep Reinforcement Learning with Intrinsic Reward Normalization","date":"2019-07-06","arxiv_id":"1907.03116","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-flappy-bird-via-asynchronous","title":"Playing Flappy Bird via Asynchronous Advantage Actor Critic Algorithm","date":"2019-07-06","arxiv_id":"1907.03098","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-modeling-chit","title":"Deep Reinforcement Learning For Modeling Chit-Chat Dialog With Discrete Attributes","date":"2019-07-05","arxiv_id":"1907.02848","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-inductive-biases-in-deep-reinforcement","title":"On Inductive Biases in Deep Reinforcement Learning","date":"2019-07-05","arxiv_id":"1907.02908","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-of-distance","title":"Self-supervised Learning of Distance Functions for Goal-Conditioned Reinforcement Learning","date":"2019-07-05","arxiv_id":"1907.02998","repositories_listed":0,"syntology":null},{"url":null,"slug":"integration-of-imitation-learning-using-gail","title":"Integration of Imitation Learning using GAIL and Reinforcement Learning using Task-achievement Rewards via Probabilistic Graphical Model","date":"2019-07-03","arxiv_id":"1907.02140","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-weaknesses-of-reinforcement-learning","title":"On the Weaknesses of Reinforcement Learning for Neural Machine Translation","date":"2019-07-03","arxiv_id":"1907.01752","repositories_listed":0,"syntology":null},{"url":null,"slug":"perspective-taking-in-deep-reinforcement","title":"Perspective Taking in Deep Reinforcement Learning Agents","date":"2019-07-03","arxiv_id":"1907.01851","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-approximate-dynamic-programming-via","title":"Safe Approximate Dynamic Programming Via Kernelized Lipschitz Estimation","date":"2019-07-03","arxiv_id":"1907.02151","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-the","title":"A Reinforcement Learning Approach for the Multichannel Rendezvous Problem","date":"2019-07-02","arxiv_id":"1907.01919","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-face-video-segmentation-via","title":"Dynamic Face Video Segmentation via Reinforcement Learning","date":"2019-07-02","arxiv_id":"1907.01296","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-from-a-few-environments-in","title":"Generalizing from a few environments in safety-critical reinforcement learning","date":"2019-07-02","arxiv_id":"1907.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"modified-actor-critics","title":"Modified Actor-Critics","date":"2019-07-02","arxiv_id":"1907.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-go-without-game-tree-search-using","title":"Playing Go without Game Tree Search Using Convolutional Neural Networks","date":"2019-07-02","arxiv_id":"1907.04658","repositories_listed":0,"syntology":null},{"url":null,"slug":"voting-based-multi-agent-reinforcement","title":"Voting-Based Multi-Agent Reinforcement Learning for Intelligent IoT","date":"2019-07-02","arxiv_id":"1907.01385","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-deep-reinforcement-learning-for","title":"Designing Deep Reinforcement Learning for Human Parameter Exploration","date":"2019-07-01","arxiv_id":"1907.00824","repositories_listed":0,"syntology":null},{"url":"/paper/end-to-end-deep-reinforcement-learning-based","slug":"end-to-end-deep-reinforcement-learning-based","title":"End-to-end Deep Reinforcement Learning Based Coreference Resolution","date":"2019-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fidi-rl-incorporating-deep-reinforcement","title":"FiDi-RL: Incorporating Deep Reinforcement Learning with Finite-Difference Policy Search for Efficient Learning of Continuous Control","date":"2019-07-01","arxiv_id":"1907.00526","repositories_listed":0,"syntology":null},{"url":null,"slug":"historical-text-normalization-with-delayed","title":"Historical Text Normalization with Delayed Rewards","date":"2019-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-world-graphs-to-accelerate","title":"Learning World Graphs to Accelerate Hierarchical Reinforcement Learning","date":"2019-07-01","arxiv_id":"1907.00664","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-semantic-similarity-as-reward-for","title":"Using Semantic Similarity as Reward for Reinforcement Learning in Sentence Generation","date":"2019-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaboration-of-ai-agents-via-cooperative","title":"Collaboration of AI Agents via Cooperative Multi-Agent Deep Reinforcement Learning","date":"2019-06-30","arxiv_id":"1907.00327","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-training-flexible-robots-using-deep","title":"On Training Flexible Robots using Deep Reinforcement Learning","date":"2019-06-29","arxiv_id":"1907.00269","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-honeypot-engagement-through","title":"Adaptive Honeypot Engagement through Reinforcement Learning of Semi-Markov Decision Processes","date":"2019-06-27","arxiv_id":"1906.12182","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-guided-deep-reinforcement","title":"Demonstration-Guided Deep Reinforcement Learning of Control Policies for Dexterous Human-Robot Interaction","date":"2019-06-27","arxiv_id":"1906.11695","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-self-tuning-regulators-to-reinforcement","title":"From self-tuning regulators to reinforcement learning and back again","date":"2019-06-27","arxiv_id":"1906.11392","repositories_listed":0,"syntology":null}],"record_sha256":"79aab077707c9c254af21119631113626123c583175c7b6076db830b31b5d272","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}