{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/106","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":106,"pages_in_order":132,"rows_per_page":100,"rows":[10501,10600],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/105","next":"/task/reinforcement-learning/papers/107","papers":[{"url":null,"slug":"resource-optimized-neural-architecture-search","title":"Resource Optimized Neural Architecture Search for 3D Medical Image Segmentation","date":"2019-09-02","arxiv_id":"1909.00548","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-reinforcement-learning-based-neural","title":"Scalable Reinforcement-Learning-Based Neural Architecture Search for Cancer Deep Learning Research","date":"2019-09-01","arxiv_id":"1909.00311","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-combine-or-not-to-combine-a-rainbow-deep","title":"To Combine or Not To Combine? A Rainbow Deep Reinforcement Learning Agent for Dialog Policies","date":"2019-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-1","title":"Deep Reinforcement Learning with Distributional Semantic Rewards for Abstractive Summarization","date":"2019-08-31","arxiv_id":"1909.00141","repositories_listed":0,"syntology":null},{"url":null,"slug":"named-entity-recognition-only-from-word","title":"Named Entity Recognition Only from Word Embeddings","date":"2019-08-31","arxiv_id":"1909.00164","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-efficiency-rl-agent","title":"Reinforcement learning with world model","date":"2019-08-30","arxiv_id":"1908.11494","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-transfer-learn","title":"Learning to Transfer Learn: Reinforcement Learning-Based Selection for Adaptive Transfer Learning","date":"2019-08-29","arxiv_id":"1908.11406","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-driven-de-novo-design","title":"PaccMann$^{RL}$: Designing anticancer drugs from transcriptomic data via reinforcement learning","date":"2019-08-29","arxiv_id":"1909.05114","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-discounted-stochastic-two-player","title":"Solving Discounted Stochastic Two-Player Games with Near-Optimal Time and Sample Complexity","date":"2019-08-29","arxiv_id":"1908.11071","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-actor-critic-reinforcement-learning-for","title":"Deep Actor-Critic Reinforcement Learning for Anomaly Detection","date":"2019-08-28","arxiv_id":"1908.10755","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-prediction-control-and","title":"Reinforcement Learning: Prediction, Control and Value Function Approximation","date":"2019-08-28","arxiv_id":"1908.10771","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-math-word-problems-with-double","title":"Solving Math Word Problems with Double-Decoder Transformer","date":"2019-08-28","arxiv_id":"1908.10924","repositories_listed":0,"syntology":null},{"url":null,"slug":"stmarl-a-spatio-temporal-multi-agent","title":"STMARL: A Spatio-Temporal Multi-Agent Reinforcement Learning Approach for Cooperative Traffic Light Control","date":"2019-08-28","arxiv_id":"1908.10577","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-data-efficient-deep-learning-approach-for","title":"A Data-Efficient Deep Learning Approach for Deployable Multimodal Social Robots","date":"2019-08-27","arxiv_id":"1908.10398","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-chatbots","title":"Deep Reinforcement Learning for Chatbots Using Clustered Actions and Human-Likeness Rewards","date":"2019-08-27","arxiv_id":"1908.10331","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-based-deep-reinforcement-learning","title":"Ensemble-Based Deep Reinforcement Learning for Chatbots","date":"2019-08-27","arxiv_id":"1908.10422","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-enhanced-politex","title":"Exploration-Enhanced POLITEX","date":"2019-08-27","arxiv_id":"1908.10479","repositories_listed":0,"syntology":null},{"url":null,"slug":"hymer-a-hybrid-machine-learning-framework-for","title":"MER-SDN: Machine Learning Framework for Traffic Aware Energy Efficient Routing in SDN","date":"2019-08-27","arxiv_id":"1909.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-to","title":"A Deep Reinforcement Learning Approach to Multi-component Job Scheduling in Edge Computing","date":"2019-08-26","arxiv_id":"1908.10290","repositories_listed":0,"syntology":null},{"url":null,"slug":"urban-flows-prediction-from-spatial-temporal","title":"Urban flows prediction from spatial-temporal data using machine learning: A survey","date":"2019-08-26","arxiv_id":"1908.10218","repositories_listed":0,"syntology":null},{"url":null,"slug":"tutorial-and-survey-on-probabilistic","title":"Tutorial and Survey on Probabilistic Graphical Model and Variational Inference in Deep Reinforcement Learning","date":"2019-08-25","arxiv_id":"1908.09381","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-action-spaces-for-learning","title":"A Comparison of Action Spaces for Learning Manipulation Tasks","date":"2019-08-23","arxiv_id":"1908.08659","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-dynamics-of-quantum-sensors","title":"Improving the dynamics of quantum sensors with reinforcement learning","date":"2019-08-22","arxiv_id":"1908.08416","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-convergence-rate-of-adaptive-multiscale","title":"On Convergence Rate of Adaptive Multiscale Value Function Approximation For Reinforcement Learning","date":"2019-08-22","arxiv_id":"1908.08578","repositories_listed":0,"syntology":null},{"url":null,"slug":"practical-risk-measures-in-reinforcement","title":"Practical Risk Measures in Reinforcement Learning","date":"2019-08-22","arxiv_id":"1908.08379","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-healthcare-a-survey","title":"Reinforcement Learning in Healthcare: A Survey","date":"2019-08-22","arxiv_id":"1908.08796","repositories_listed":0,"syntology":null},{"url":null,"slug":"190807795","title":"Dialog State Tracking with Reinforced Data Augmentation","date":"2019-08-21","arxiv_id":"1908.07795","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-cyber-physical-systems-from-the","title":"Analyzing Cyber-Physical Systems from the Perspective of Artificial Intelligence","date":"2019-08-21","arxiv_id":"1908.11779","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-foreign","title":"Deep Reinforcement Learning for Foreign Exchange Trading","date":"2019-08-21","arxiv_id":"1908.08036","repositories_listed":0,"syntology":null},{"url":null,"slug":"190807617","title":"Reinforcement Learning is not a Causal problem","date":"2019-08-20","arxiv_id":"1908.07617","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-actor-critic-reinforcement-learning","title":"A Deep Actor-Critic Reinforcement Learning Framework for Dynamic Multichannel Access","date":"2019-08-20","arxiv_id":"1908.08401","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-sit-synthesizing-human-chair","title":"Learning to Sit: Synthesizing Human-Chair Interactions via Hierarchical Control","date":"2019-08-20","arxiv_id":"1908.07423","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-intrinsic-motivation-in","title":"A survey on intrinsic motivation in reinforcement learning","date":"2019-08-19","arxiv_id":"1908.06976","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-flight-control-a-domain","title":"A Domain-Knowledge-Aided Deep Reinforcement Learning Approach for Flight Control Design","date":"2019-08-19","arxiv_id":"1908.06884","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-advertise-for-organic-traffic","title":"Learning to Advertise for Organic Traffic Maximization in E-Commerce Product Feeds","date":"2019-08-19","arxiv_id":"1908.06698","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-multi-stage-cascading-failure-by","title":"Mitigating Multi-Stage Cascading Failure by Reinforcement Learning","date":"2019-08-19","arxiv_id":"1908.06599","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-applications","title":"Reinforcement Learning Applications","date":"2019-08-19","arxiv_id":"1908.06973","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-in-deep-reinforcement-learning-using-2","title":"Transfer in Deep Reinforcement Learning using Knowledge Graphs","date":"2019-08-19","arxiv_id":"1908.06556","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-update-and-unified-representation","title":"Iterative Update and Unified Representation for Multi-Agent Reinforcement Learning","date":"2019-08-16","arxiv_id":"1908.06758","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-feature-selection-for-activity","title":"Online Feature Selection for Activity Recognition using Reinforcement Learning with Multiple Feedback","date":"2019-08-16","arxiv_id":"1908.06134","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-lookahead-reinforcement-learning","title":"Model-based Lookahead Reinforcement Learning","date":"2019-08-15","arxiv_id":"1908.06012","repositories_listed":0,"syntology":null},{"url":null,"slug":"playing-a-strategy-game-with-knowledge-based","title":"Playing a Strategy Game with Knowledge-Based Reinforcement Learning","date":"2019-08-15","arxiv_id":"1908.05472","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-deep-reinforcement-learning-4","title":"Sample-efficient Deep Reinforcement Learning with Imaginary Rollouts for Human-Robot Interaction","date":"2019-08-15","arxiv_id":"1908.05546","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-end-to-end-learning-for-efficient","title":"Towards End-to-End Learning for Efficient Dialogue Agent by Modeling Looking-ahead Ability","date":"2019-08-15","arxiv_id":"1908.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"unpaired-cross-lingual-image-caption","title":"Unpaired Cross-lingual Image Caption Generation with Self-Supervised Rewards","date":"2019-08-15","arxiv_id":"1908.05407","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-for-high-dimensional-state","title":"Continuous Control for High-Dimensional State Spaces: An Interactive Learning Approach","date":"2019-08-14","arxiv_id":"1908.05256","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-transfer-in-deep-reinforcement-learning","title":"Skill Transfer in Deep Reinforcement Learning under Morphological Heterogeneity","date":"2019-08-14","arxiv_id":"1908.05265","repositories_listed":0,"syntology":null},{"url":null,"slug":"competitive-multi-agent-deep-reinforcement","title":"Competitive Multi-Agent Deep Reinforcement Learning with Counterfactual Thinking","date":"2019-08-13","arxiv_id":"1908.04573","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-crystallized-adaptivity-to-fluid","title":"From Crystallized Adaptivity to Fluid Adaptivity in Deep Reinforcement Learning -- Insights from Biological Systems on Adaptive Flexibility","date":"2019-08-13","arxiv_id":"1908.05348","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-manipulation-via-locomotion-using","title":"Multi-Agent Manipulation via Locomotion using Hierarchical Sim2Real","date":"2019-08-13","arxiv_id":"1908.05224","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-tampering-problems-and-solutions-in","title":"Reward Tampering Problems and Solutions in Reinforcement Learning: A Causal Influence Diagram Perspective","date":"2019-08-13","arxiv_id":"1908.04734","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-adaptation-with-meta-reinforcement","title":"Fast Adaptation with Meta-Reinforcement Learning for Trust Modelling in Human-Robot Interaction","date":"2019-08-12","arxiv_id":"1908.04087","repositories_listed":0,"syntology":null},{"url":null,"slug":"superstition-in-the-network-deep","title":"Superstition in the Network: Deep Reinforcement Learning Plays Deceptive Games","date":"2019-08-12","arxiv_id":"1908.04436","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-cooperative-multi-agent-deep","title":"A Review of Cooperative Multi-Agent Deep Reinforcement Learning","date":"2019-08-11","arxiv_id":"1908.03963","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-traffic-signal-control-using-a","title":"Large-Scale Traffic Signal Control Using a Novel Multi-Agent Reinforcement Learning","date":"2019-08-10","arxiv_id":"1908.03761","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-in-motion-and-interaction","title":"Learning to Explore in Motion and Interaction Tasks","date":"2019-08-10","arxiv_id":"1908.03731","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservatives-overfit-liberals-underfit-the","title":"\"Conservatives Overfit, Liberals Underfit\": The Social-Psychological Control of Affect and Uncertainty","date":"2019-08-08","arxiv_id":"1908.03106","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-reinforcement-learning-a-new","title":"Incremental Reinforcement Learning --- a New Continuous Reinforcement Learning Frame Based on Stochastic Differential Equation methods","date":"2019-08-08","arxiv_id":"1908.02974","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-grasp-from-25d-images-a-deep","title":"Learning to Grasp from 2.5D images: a Deep Reinforcement Learning Approach","date":"2019-08-08","arxiv_id":"1908.03440","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-relation-learning-for-group","title":"Progressive Relation Learning for Group Activity Recognition","date":"2019-08-08","arxiv_id":"1908.02948","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-wise-control-variates-for-variance","title":"Trajectory-wise Control Variates for Variance Reduction in Policy Gradient Methods","date":"2019-08-08","arxiv_id":"1908.03263","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-multi-agent-temporal-difference-learning","title":"Fast Multi-Agent Temporal-Difference Learning via Homotopy Stochastic Primal-Dual Optimization","date":"2019-08-07","arxiv_id":"1908.02805","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-traffic-signal-control-using","title":"Large-scale traffic signal control using machine learning: some traffic flow considerations","date":"2019-08-07","arxiv_id":"1908.02673","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-optimal-sequencing-of","title":"Task-Oriented Optimal Sequencing of Visualization Charts","date":"2019-08-07","arxiv_id":"1908.02502","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-stochastic-game-theory-approach-for-the","title":"A physics-informed reinforcement learning approach for the interfacial area transport in two-phase flow","date":"2019-08-06","arxiv_id":"1908.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"age-of-information-aware-radio-resource","title":"Age of Information-Aware Radio Resource Management in Vehicular Networks: A Proactive Deep Reinforcement Learning Perspective","date":"2019-08-06","arxiv_id":"1908.02047","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-recurrent-q-learning-for-backchannel","title":"Batch Recurrent Q-Learning for Backchannel Generation Towards Engaging Agents","date":"2019-08-06","arxiv_id":"1908.02037","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-bonus-based-exploration-methods","title":"Benchmarking Bonus-Based Exploration Methods on the Arcade Learning Environment","date":"2019-08-06","arxiv_id":"1908.02388","repositories_listed":0,"syntology":null},{"url":null,"slug":"promoting-coordination-through-policy","title":"Promoting Coordination through Policy Regularization in Multi-Agent Deep Reinforcement Learning","date":"2019-08-06","arxiv_id":"1908.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-control-with-metric-learning","title":"Attention Control with Metric Learning Alignment for Image Set-based Recognition","date":"2019-08-05","arxiv_id":"1908.01872","repositories_listed":0,"syntology":null},{"url":null,"slug":"construction-of-macro-actions-for-deep","title":"Reusability and Transferability of Macro Actions for Reinforcement Learning","date":"2019-08-05","arxiv_id":"1908.01478","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-driven-backchannel-generation-using","title":"Speech Driven Backchannel Generation using Deep Q-Network for Enhancing Engagement in Human-Robot Interaction","date":"2019-08-05","arxiv_id":"1908.01618","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-system","title":"A View on Deep Reinforcement Learning in System Optimization","date":"2019-08-04","arxiv_id":"1908.01275","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-training-and-design-of-photonic","slug":"efficient-training-and-design-of-photonic","title":"Efficient training and design of photonic neural network through neuroevolution","date":"2019-08-04","arxiv_id":"1908.08012","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-with-reward","title":"Adaptive Stress Testing with Reward Augmentation for Autonomous Vehicle Validation","date":"2019-08-02","arxiv_id":"1908.01046","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-deep-reinforcement-learning-in","title":"Improving Deep Reinforcement Learning in Minecraft with Action Advice","date":"2019-08-02","arxiv_id":"1908.01007","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-reinforcement-learning-for","title":"Curiosity-driven Reinforcement Learning for Diverse Visual Paragraph Generation","date":"2019-08-01","arxiv_id":"1908.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-when-to-drive-in-intersections-by","title":"Learning When to Drive in Intersections by Combining Reinforcement Learning and Model Predictive Control","date":"2019-08-01","arxiv_id":"1908.00177","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-simplex-architecture","title":"Neural Simplex Architecture","date":"2019-08-01","arxiv_id":"1908.00528","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimality-and-approximation-with-policy","title":"On the Theory of Policy Gradient Methods: Optimality, Approximation, and Distribution Shift","date":"2019-08-01","arxiv_id":"1908.00261","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-personalized","title":"Reinforcement Learning for Personalized Dialogue Management","date":"2019-08-01","arxiv_id":"1908.00286","repositories_listed":0,"syntology":null},{"url":null,"slug":"robby-is-not-a-robber-anymore-on-the-use-of","title":"Robby is Not a Robber (anymore): On the Use of Institutions for Learning Normative Behavior","date":"2019-08-01","arxiv_id":"1908.02138","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-multiple","title":"Inverse Reinforcement Learning with Multiple Ranked Experts","date":"2019-07-31","arxiv_id":"1907.13411","repositories_listed":0,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-based","slug":"multi-agent-reinforcement-learning-based","title":"Multi-Agent Reinforcement Learning Based Frame Sampling for Effective Untrimmed Video Recognition","date":"2019-07-31","arxiv_id":"1907.13369","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-point-bandit-algorithms-for","title":"Multi-Point Bandit Algorithms for Nonstationary Online Nonconvex Optimization","date":"2019-07-31","arxiv_id":"1907.13616","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-attacks-on-reinforcement-learning","title":"Optimal Attacks on Reinforcement Learning Policies","date":"2019-07-31","arxiv_id":"1907.13548","repositories_listed":0,"syntology":null},{"url":null,"slug":"precodernet-hybrid-beamforming-for-millimeter","title":"PrecoderNet: Hybrid Beamforming for Millimeter Wave Systems with Deep Reinforcement Learning","date":"2019-07-31","arxiv_id":"1907.13266","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepplace-learning-to-place-applications-in","title":"DeepPlace: Learning to Place Applications in Multi-Tenant Clusters","date":"2019-07-30","arxiv_id":"1907.12916","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-unsupervised-learning-for","title":"Model-Free Unsupervised Learning for Optimization Problems with Constraints","date":"2019-07-30","arxiv_id":"1907.12706","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-robust-reinforcement-learning","title":"Wasserstein Robust Reinforcement Learning","date":"2019-07-30","arxiv_id":"1907.13196","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-driven-sequential-data-abstraction","title":"Goal-Driven Sequential Data Abstraction","date":"2019-07-29","arxiv_id":"1907.12336","repositories_listed":0,"syntology":null},{"url":null,"slug":"lotka-volterra-competition-mechanism-embedded","title":"Lotka-Volterra competition mechanism embedded in a decision-making method","date":"2019-07-29","arxiv_id":"1907.12399","repositories_listed":0,"syntology":null},{"url":null,"slug":"particle-swarm-optimisation-for-evolving-deep","title":"Particle Swarm Optimisation for Evolving Deep Neural Networks for Image Classification by Evolving and Stacking Transferable Blocks","date":"2019-07-29","arxiv_id":"1907.12659","repositories_listed":0,"syntology":null},{"url":null,"slug":"taxable-stock-trading-with-deep-reinforcement","title":"Taxable Stock Trading with Deep Reinforcement Learning","date":"2019-07-28","arxiv_id":"1907.12093","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-should-i-ask-using-conversationally-1","title":"What Should I Ask? Using Conversationally Informative Rewards for Goal-Oriented Visual Dialog","date":"2019-07-28","arxiv_id":"1907.12021","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-mind-defeating-stealthy-dos-attacks-in-sdn","title":"Q-MIND: Defeating Stealthy DoS Attacks in SDN with a Machine-learning based Defense Framework","date":"2019-07-27","arxiv_id":"1907.11887","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-bellman-optimality-principle","title":"A Unified Bellman Optimality Principle Combining Reward Maximization and Empowerment","date":"2019-07-26","arxiv_id":"1907.12392","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-on-line-learning","title":"An Information-theoretic On-line Learning Principle for Specialization in Hierarchical Decision-Making Systems","date":"2019-07-26","arxiv_id":"1907.11452","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-personalized","title":"Deep Reinforcement Learning for Personalized Search Story Recommendation","date":"2019-07-26","arxiv_id":"1907.11754","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-continuous-time-mean-variance","title":"Large scale continuous-time mean-variance portfolio allocation via reinforcement learning","date":"2019-07-26","arxiv_id":"1907.11718","repositories_listed":0,"syntology":null}],"record_sha256":"635ff5733a6b1bcd0d97b08a8fb963104ea5b72a60bc3dd5717ae75166370f86","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}