{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/108","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":108,"pages_in_order":132,"rows_per_page":100,"rows":[10701,10800],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/107","next":"/task/reinforcement-learning/papers/109","papers":[{"url":null,"slug":"what-should-i-ask-using-conversationally","title":"What Should I Ask? Using Conversationally Informative Rewards for Goal-oriented Visual Dialog.","date":"2019-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaboration-of-ai-agents-via-cooperative","title":"Collaboration of AI Agents via Cooperative Multi-Agent Deep Reinforcement Learning","date":"2019-06-30","arxiv_id":"1907.00327","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-for-intelligent","title":"Machine Learning for Intelligent Authentication in 5G-and-Beyond Wireless Networks","date":"2019-06-30","arxiv_id":"1907.00429","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-fairness","title":"Multi-Armed Bandits with Fairness Constraints for Distributing Resources to Human Teammates","date":"2019-06-30","arxiv_id":"1907.00313","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-training-flexible-robots-using-deep","title":"On Training Flexible Robots using Deep Reinforcement Learning","date":"2019-06-29","arxiv_id":"1907.00269","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-cope-with-adversarial-attacks","title":"Learning to Cope with Adversarial Attacks","date":"2019-06-28","arxiv_id":"1906.12061","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-honeypot-engagement-through","title":"Adaptive Honeypot Engagement through Reinforcement Learning of Semi-Markov Decision Processes","date":"2019-06-27","arxiv_id":"1906.12182","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-guided-deep-reinforcement","title":"Demonstration-Guided Deep Reinforcement Learning of Control Policies for Dexterous Human-Robot Interaction","date":"2019-06-27","arxiv_id":"1906.11695","repositories_listed":0,"syntology":null},{"url":null,"slug":"extra-transfer-guided-exploration","title":"ExTra: Transfer-guided Exploration","date":"2019-06-27","arxiv_id":"1906.11785","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-self-tuning-regulators-to-reinforcement","title":"From self-tuning regulators to reinforcement learning and back again","date":"2019-06-27","arxiv_id":"1906.11392","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantile-regression-deep-reinforcement","title":"Learning Policies through Quantile Regression","date":"2019-06-27","arxiv_id":"1906.11941","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-simulating-environments-in","title":"Toward Simulating Environments in Reinforcement Learning Based Recommendations","date":"2019-06-27","arxiv_id":"1906.11462","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tractable-algorithm-for-finite-horizon","title":"A Tractable Algorithm For Finite-Horizon Continuous Reinforcement Learning","date":"2019-06-26","arxiv_id":"1906.11245","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-dynamic-programming-for-linear","title":"Approximate Dynamic Programming For Linear Systems with State and Input Constraints","date":"2019-06-26","arxiv_id":"1906.11369","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-navigation-of-active-particles-in","title":"Efficient Navigation of Colloidal Robots in an Unknown Environment via Deep Reinforcement Learning","date":"2019-06-26","arxiv_id":"1906.10844","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-hierarchical-policies-for","title":"Compositional Transfer in Hierarchical Reinforcement Learning","date":"2019-06-26","arxiv_id":"1906.11228","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-formal-models-of-partially","title":"Rethinking Formal Models of Partially Observable Multiagent Decision Making","date":"2019-06-26","arxiv_id":"1906.11110","repositories_listed":0,"syntology":null},{"url":null,"slug":"expected-sarsa-with-control-variate-for","title":"Expected Sarsa($λ$) with Control Variate for Variance Reduction","date":"2019-06-25","arxiv_id":"1906.11058","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-causal-state-representations-of","title":"Learning Causal State Representations of Partially Observable Environments","date":"2019-06-25","arxiv_id":"1906.10437","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-proximaltrust-region-policy","title":"Neural Proximal/Trust Region Policy Optimization Attains Globally Optimal Policy","date":"2019-06-25","arxiv_id":"1906.10306","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-multi-agent-learning-in-team-sports-games","title":"On Multi-Agent Learning in Team Sports Games","date":"2019-06-25","arxiv_id":"1906.10124","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-proximal-policy-optimization","title":"Optimistic Proximal Policy Optimization","date":"2019-06-25","arxiv_id":"1906.11075","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-with-stochastic-mirror","title":"Policy Optimization with Stochastic Mirror Descent","date":"2019-06-25","arxiv_id":"1906.10462","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-competitive","title":"Reinforcement Learning with Competitive Ensembles of Information-Constrained Primitives","date":"2019-06-25","arxiv_id":"1906.10667","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-model-based-policy","title":"Uncertainty-aware Model-based Policy Optimization","date":"2019-06-25","arxiv_id":"1906.10717","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theoretical-connection-between-statistical","title":"A Theoretical Connection Between Statistical Physics and Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.10228","repositories_listed":0,"syntology":null},{"url":null,"slug":"deceptive-reinforcement-learning-under","title":"Deceptive Reinforcement Learning Under Adversarial Manipulations on Cost Signals","date":"2019-06-24","arxiv_id":"1906.10571","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-conservative-policy-iteration","title":"Deep Conservative Policy Iteration","date":"2019-06-24","arxiv_id":"1906.09784","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-driven-models","title":"Event-Driven Models","date":"2019-06-24","arxiv_id":"1906.10740","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-conditioned-on","title":"Inverse reinforcement learning conditioned on brain scan","date":"2019-06-24","arxiv_id":"1906.09770","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-use-of-experience-in-first-person","title":"Optimal Use of Experience in First Person Shooter Environments","date":"2019-06-24","arxiv_id":"1906.09734","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-networks-with-motivation","title":"Neural networks with motivation","date":"2019-06-23","arxiv_id":"1906.09528","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-feasibility-of-learning-rather-than","title":"On the Feasibility of Learning, Rather than Assuming, Human Biases for Reward Inference","date":"2019-06-23","arxiv_id":"1906.09624","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-trajectory","title":"Reinforcement Learning-Based Trajectory Design for the Aerial Base Stations","date":"2019-06-23","arxiv_id":"1906.09550","repositories_listed":0,"syntology":null},{"url":null,"slug":"rltm-an-efficient-neural-ir-framework-for","title":"RLTM: An Efficient Neural IR Framework for Long Documents","date":"2019-06-22","arxiv_id":"1906.09404","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-state-aliasing-in-structured","title":"A Study of State Aliasing in Structured Prediction with RNNs","date":"2019-06-21","arxiv_id":"1906.09310","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-reinforcement-learning-with","title":"Continual Reinforcement Learning with Diversity Exploration and Adversarial Self-Correction","date":"2019-06-21","arxiv_id":"1906.09205","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-skill-embeddings-for","title":"Disentangled Skill Embeddings for Reinforcement Learning","date":"2019-06-21","arxiv_id":"1906.09223","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-reinforcement-learning-techniques","title":"Leveraging Reinforcement Learning Techniques for Effective Policy Adoption and Validation","date":"2019-06-21","arxiv_id":"1906.09340","repositories_listed":0,"syntology":null},{"url":null,"slug":"revised-progressive-hedging-algorithm-based","title":"Revised Progressive-Hedging-Algorithm Based Two-layer Solution Scheme for Bayesian Reinforcement Learning","date":"2019-06-21","arxiv_id":"1906.09035","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-belief-states-with-generative","title":"Shaping Belief States with Generative Environment Models for RL","date":"2019-06-21","arxiv_id":"1906.09237","repositories_listed":0,"syntology":null},{"url":null,"slug":"cache-aided-noma-mobile-edge-computing-a","title":"Cache-Aided NOMA Mobile Edge Computing: A Reinforcement Learning Approach","date":"2019-06-20","arxiv_id":"1906.08812","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-lane-changing-via-deep","title":"Cooperative Lane Changing via Deep Reinforcement Learning","date":"2019-06-20","arxiv_id":"1906.08662","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-needles-in-a-moving-haystack","title":"Finding Needles in a Moving Haystack: Prioritizing Alerts with Adversarial Reinforcement Learning","date":"2019-06-20","arxiv_id":"1906.08805","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-finite-horizon-two-armed-bandit-problem","title":"The Finite-Horizon Two-Armed Bandit Problem with Binary Responses: A Multidisciplinary Survey of the History, State of the Art, and Myths","date":"2019-06-20","arxiv_id":"1906.10173","repositories_listed":0,"syntology":null},{"url":null,"slug":"variable-impedance-control-in-end-effector","title":"Variable Impedance Control in End-Effector Space: An Action Space for Reinforcement Learning in Contact-Rich Tasks","date":"2019-06-20","arxiv_id":"1906.08880","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-multiple-agents-learn-to-schedule-a","title":"When Multiple Agents Learn to Schedule: A Distributed Radio Resource Management Framework","date":"2019-06-20","arxiv_id":"1906.08792","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-behaviour-via-intrinsic-reward-a","title":"Adapting Behaviour via Intrinsic Reward: A Survey and Empirical Study","date":"2019-06-19","arxiv_id":"1906.07865","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-temporal-difference-learning-for","title":"Adaptive Temporal-Difference Learning for Policy Evaluation with Per-State Uncertainty Estimates","date":"2019-06-19","arxiv_id":"1906.07987","repositories_listed":0,"syntology":null},{"url":null,"slug":"experience-replay-optimization","title":"Experience Replay Optimization","date":"2019-06-19","arxiv_id":"1906.08387","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-convergence-of-policy-gradient-methods-3","title":"Global Convergence of Policy Gradient Methods to (Almost) Locally Optimal Policies","date":"2019-06-19","arxiv_id":"1906.08383","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-user-resource-control-with-deep","title":"Multi-user Resource Control with Deep Reinforcement Learning in IoT Edge Computing","date":"2019-06-19","arxiv_id":"1906.07860","repositories_listed":0,"syntology":null},{"url":null,"slug":"qxplore-q-learning-exploration-by-maximizing","title":"Reward Prediction Error as an Exploration Objective in Deep RL","date":"2019-06-19","arxiv_id":"1906.08189","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-adversarial-imitation-learning","title":"Wasserstein Adversarial Imitation Learning","date":"2019-06-19","arxiv_id":"1906.08113","repositories_listed":0,"syntology":null},{"url":null,"slug":"directed-exploration-for-reinforcement","title":"Directed Exploration for Reinforcement Learning","date":"2019-06-18","arxiv_id":"1906.07805","repositories_listed":0,"syntology":null},{"url":null,"slug":"disco-influence-maximization-meets-network","title":"DISCO: Influence Maximization Meets Network Embedding and Deep Learning","date":"2019-06-18","arxiv_id":"1906.07378","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-for","title":"Evolutionary Reinforcement Learning for Sample-Efficient Multiagent Coordination","date":"2019-06-18","arxiv_id":"1906.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"gap-increasing-policy-evaluation-for","title":"Gap-Increasing Policy Evaluation for Efficient and Noise-Tolerant Reinforcement Learning","date":"2019-06-18","arxiv_id":"1906.07586","repositories_listed":0,"syntology":null},{"url":null,"slug":"hill-climbing-on-value-estimates-for-search","title":"Hill Climbing on Value Estimates for Search-control in Dyna","date":"2019-06-18","arxiv_id":"1906.07791","repositories_listed":0,"syntology":null},{"url":null,"slug":"ridm-reinforced-inverse-dynamics-modeling-for","title":"RIDM: Reinforced Inverse Dynamics Modeling for Learning from a Single Observed Demonstration","date":"2019-06-18","arxiv_id":"1906.07372","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-for-continuous","title":"Robust Reinforcement Learning for Continuous Control with Model Misspecification","date":"2019-06-18","arxiv_id":"1906.07516","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-adversarial-imitation","title":"Sample-efficient Adversarial Imitation Learning from Observation","date":"2019-06-18","arxiv_id":"1906.07374","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-white-box-benchmarks-for-algorithm","title":"Towards White-box Benchmarks for Algorithm Control","date":"2019-06-18","arxiv_id":"1906.07644","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gray-box-approach-for-curriculum-learning","title":"A gray-box approach for curriculum learning","date":"2019-06-17","arxiv_id":"1906.06812","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-optimization-with-binary-auxiliary","title":"Bayesian Optimization with Binary Auxiliary Information","date":"2019-06-17","arxiv_id":"1906.07277","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-model-based-reinforcement-learning","title":"Iterative Model-Based Reinforcement Learning Using Simulations in the Differentiable Neural Computer","date":"2019-06-17","arxiv_id":"1906.07248","repositories_listed":0,"syntology":null},{"url":null,"slug":"lpaintb-learning-to-paint-from-self","title":"LPaintB: Learning to Paint from Self-Supervision","date":"2019-06-17","arxiv_id":"1906.06841","repositories_listed":0,"syntology":null},{"url":null,"slug":"pacman-a-planner-actor-critic-architecture","title":"A Joint Planning and Learning Framework for Human-Aided Decision-Making","date":"2019-06-17","arxiv_id":"1906.07268","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-successor-features-based-deep","title":"Universal Successor Features Based Deep Reinforcement Learning for Navigation","date":"2019-06-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-driven-heuristic","title":"Reinforcement Learning Driven Heuristic Optimization","date":"2019-06-16","arxiv_id":"1906.06639","repositories_listed":0,"syntology":null},{"url":null,"slug":"injecting-prior-knowledge-for-transfer","title":"Injecting Prior Knowledge for Transfer Learning into Reinforcement Learning Algorithms using Logic Tensor Networks","date":"2019-06-15","arxiv_id":"1906.06576","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-non-uniform-state","title":"Reinforcement Learning with Non-uniform State Representations for Adaptive Search","date":"2019-06-15","arxiv_id":"1906.06588","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-policy-gradients-direct-optimization","title":"Direct Policy Gradients: Direct Optimization of Policies in Discrete Action Spaces","date":"2019-06-14","arxiv_id":"1906.06062","repositories_listed":0,"syntology":null},{"url":null,"slug":"epistemic-risk-sensitive-reinforcement","title":"Epistemic Risk-Sensitive Reinforcement Learning","date":"2019-06-14","arxiv_id":"1906.06273","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-q-learning-with-function","title":"Provably Efficient $Q$-learning with Function Approximation via Distribution Shift Error Checking Oracle","date":"2019-06-14","arxiv_id":"1906.06321","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-tuning-sectorization-deep-reinforcement","title":"Self-Tuning Sectorization: Deep Reinforcement Learning Meets Broadcast Beam Optimization","date":"2019-06-14","arxiv_id":"1906.06021","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-cyber","title":"Deep Reinforcement Learning for Cyber Security","date":"2019-06-13","arxiv_id":"1906.05799","repositories_listed":0,"syntology":null},{"url":null,"slug":"early-detection-of-long-term-evaluation","title":"Early Detection of Long Term Evaluation Criteria in Online Controlled Experiments","date":"2019-06-13","arxiv_id":"1906.05959","repositories_listed":0,"syntology":null},{"url":null,"slug":"jacobian-policy-optimizations","title":"Conditioning of Reinforcement Learning Agents and its Policy Regularization Application","date":"2019-06-13","arxiv_id":"1906.05437","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-and-interpreting-real-world-human","title":"Modeling and Interpreting Real-world Human Risk Decision Making with Inverse Reinforcement Learning","date":"2019-06-13","arxiv_id":"1906.05803","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-policy-adaptation-for-hierarchical","title":"Sub-policy Adaptation for Hierarchical Reinforcement Learning","date":"2019-06-13","arxiv_id":"1906.05862","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-unmanned","title":"Deep Reinforcement Learning for Unmanned Aerial Vehicle-Assisted Vehicular Networks","date":"2019-06-12","arxiv_id":"1906.05015","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-minimization-for-reinforcement","title":"Regret Minimization for Reinforcement Learning by Evaluating the Optimal Bias Function","date":"2019-06-12","arxiv_id":"1906.05110","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-adaptive-optimal","title":"Adaptive Optimal Control for Reference Tracking Independent of Exo-System Dynamics","date":"2019-06-12","arxiv_id":"1906.05085","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-goal-trees-a-framework-for-goal-directed","title":"Sub-Goal Trees -- a Framework for Goal-Directed Trajectory Prediction and Optimization","date":"2019-06-12","arxiv_id":"1906.05329","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-approach-between-adversarial","title":"A Hybrid Approach Between Adversarial Generative Networks and Actor-Critic Policy Gradient for Low Rate High-Resolution Image Compression","date":"2019-06-11","arxiv_id":"1906.04681","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-reinforcement-learning-deployed-in","title":"Continual Reinforcement Learning deployed in Real-life using Policy Distillation and Sim2Real Transfer","date":"2019-06-11","arxiv_id":"1906.04452","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealing-with-non-stationarity-in-multi-agent","title":"Dealing with Non-Stationarity in Multi-Agent Deep Reinforcement Learning","date":"2019-06-11","arxiv_id":"1906.04737","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-control-of-artificial-avatars","title":"Deep learning control of artificial avatars in group coordination tasks","date":"2019-06-11","arxiv_id":"1906.04656","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-integer","title":"Reinforcement Learning for Integer Programming: Learning to Cut","date":"2019-06-11","arxiv_id":"1906.04859","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-minimalist-numeral","title":"Reinforcement Learning of Minimalist Numeral Grammars","date":"2019-06-11","arxiv_id":"1906.04447","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-inverse-reinforcement-learning-for","title":"Towards Inverse Reinforcement Learning for Limit Order Book Dynamics","date":"2019-06-11","arxiv_id":"1906.04813","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-reinforcement-learning-informed","title":"A Survey of Reinforcement Learning Informed by Natural Language","date":"2019-06-10","arxiv_id":"1906.03926","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacking-graph-convolutional-networks-via","title":"Attacking Graph Convolutional Networks via Rewiring","date":"2019-06-10","arxiv_id":"1906.03750","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-discrete","title":"Deep Reinforcement Learning with Discrete Normalized Advantage Functions for Resource Management in Network Slicing","date":"2019-06-10","arxiv_id":"1906.04594","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-optimality-of-sparse-model-based","title":"Model-Based Reinforcement Learning with a Generative Model is Minimax Optimal","date":"2019-06-10","arxiv_id":"1906.03804","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-heterogeneous-scheduler","title":"Neural Heterogeneous Scheduler","date":"2019-06-09","arxiv_id":"1906.03724","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-by-modeling-a-distribution","title":"Transfer Learning by Modeling a Distribution over Policies","date":"2019-06-09","arxiv_id":"1906.03574","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-off-policy-evaluation-for","title":"Towards Optimal Off-Policy Evaluation for Reinforcement Learning with Marginalized Importance Sampling","date":"2019-06-08","arxiv_id":"1906.03393","repositories_listed":0,"syntology":null},{"url":null,"slug":"figure-captioning-with-reasoning-and-sequence","title":"Figure Captioning with Reasoning and Sequence-Level Training","date":"2019-06-07","arxiv_id":"1906.02850","repositories_listed":0,"syntology":null}],"record_sha256":"d8fcafa70bf9c94f1e943ba5e0400a185c9b0763a20755322e67871da0494424","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}