{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/142","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":142,"pages_in_order":152,"rows_per_page":100,"rows":[14101,14200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/141","next":"/task/reinforcement-learning-1/papers/143","papers":[{"url":null,"slug":"reinforcement-learning-using-augmented-neural","title":"Reinforcement Learning using Augmented Neural Networks","date":"2018-06-20","arxiv_id":"1806.07692","repositories_listed":0,"syntology":null},{"url":null,"slug":"skilled-experience-catalogue-a-skill","title":"Skilled Experience Catalogue: A Skill-Balancing Mechanism for Non-Player Characters using Reinforcement Learning","date":"2018-06-20","arxiv_id":"1806.07637","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-inverse-reinforcement-learning","title":"A Survey of Inverse Reinforcement Learning: Challenges, Methods and Progress","date":"2018-06-18","arxiv_id":"1806.06877","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-strategy-for-implementing-curiosity","title":"A unified strategy for implementing curiosity and empowerment driven reinforcement learning","date":"2018-06-18","arxiv_id":"1806.06505","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-outside-the-viability-kernel","title":"Learning from Outside the Viability Kernel: Why we Should Build Robots that can Fall with Grace","date":"2018-06-18","arxiv_id":"1806.06569","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-policy-representations-in-multiagent","title":"Learning Policy Representations in Multiagent Systems","date":"2018-06-17","arxiv_id":"1806.06464","repositories_listed":0,"syntology":null},{"url":null,"slug":"handling-cold-start-collaborative-filtering","title":"Handling Cold-Start Collaborative Filtering with Reinforcement Learning","date":"2018-06-16","arxiv_id":"1806.06192","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-online-prediction-algorithm-for","title":"An Online Prediction Algorithm for Reinforcement Learning with Linear Function Approximation using Cross Entropy Method","date":"2018-06-15","arxiv_id":"1806.06720","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-width-based-planning-with-compact","title":"Improving width-based planning with compact policies","date":"2018-06-15","arxiv_id":"1806.05898","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-policy-and-reward-reinforcement","title":"Multi-Level Policy and Reward Reinforcement Learning for Image Captioning","date":"2018-06-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-shooting-for-bots-in-first-person","title":"Adaptive Shooting for Bots in First Person Shooter Games Using Reinforcement Learning","date":"2018-06-14","arxiv_id":"1806.05554","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dynamic-urban","title":"Deep Reinforcement Learning for Dynamic Urban Transportation Problems","date":"2018-06-14","arxiv_id":"1806.05310","repositories_listed":0,"syntology":null},{"url":null,"slug":"qualitative-measurements-of-policy","title":"Qualitative Measurements of Policy Discrepancy for Return-Based Deep Q-Network","date":"2018-06-14","arxiv_id":"1806.06953","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-shoot-in-first-person-shooter","title":"Learning to Shoot in First Person Shooter Games by Stabilizing Actions and Clustering Rewards for Reinforcement Learning","date":"2018-06-13","arxiv_id":"1806.05117","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-function-valued","title":"Reinforcement Learning with Function-Valued Action Spaces for Partial Differential Equation Control","date":"2018-06-13","arxiv_id":"1806.06931","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-transferable-active-learning","title":"Meta-Learning Transferable Active Learning Policies by Deep Reinforcement Learning","date":"2018-06-12","arxiv_id":"1806.04798","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-with-1","title":"Multi-Agent Deep Reinforcement Learning with Human Strategies","date":"2018-06-12","arxiv_id":"1806.04562","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-meta-learning-for-reinforcement","title":"Unsupervised Meta-Learning for Reinforcement Learning","date":"2018-06-12","arxiv_id":"1806.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-generalized-bellman-update-for","title":"An Efficient, Generalized Bellman Update For Cooperative Inverse Reinforcement Learning","date":"2018-06-11","arxiv_id":"1806.03820","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-curiosity-loops-in-social-environments","title":"Deep Curiosity Loops in Social Environments","date":"2018-06-10","arxiv_id":"1806.03645","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-policy-for-reinforcement-learning","title":"Implicit Policy for Reinforcement Learning","date":"2018-06-10","arxiv_id":"1806.06798","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-view-planning-with-multi-scale-deep","title":"Automatic View Planning with Multi-scale Deep Reinforcement Learning Agents","date":"2018-06-08","arxiv_id":"1806.03228","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-value-function-approximation","title":"Continuous-time Value Function Approximation in Reproducing Kernel Hilbert Spaces","date":"2018-06-08","arxiv_id":"1806.02985","repositories_listed":0,"syntology":null},{"url":null,"slug":"program-synthesis-through-reinforcement","title":"Program Synthesis Through Reinforcement Learning Guided Tree Search","date":"2018-06-08","arxiv_id":"1806.02932","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-consistent-trajectory-autoencoder","title":"Self-Consistent Trajectory Autoencoder: Hierarchical Reinforcement Learning with Trajectory Embeddings","date":"2018-06-07","arxiv_id":"1806.02813","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-and-removing-exogenous-state","title":"Discovering and Removing Exogenous State Variables and Rewards for Reinforcement Learning","date":"2018-06-05","arxiv_id":"1806.01584","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixmatch-agent-curricula-for-reinforcement","title":"Mix&Match - Agent Curricula for Reinforcement Learning","date":"2018-06-05","arxiv_id":"1806.01780","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-framework","title":"Adversarial Reinforcement Learning Framework for Benchmarking Collision Avoidance Mechanisms in Autonomous Vehicles","date":"2018-06-04","arxiv_id":"1806.01368","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigation-of-policy-manipulation-attacks-on","title":"Mitigation of Policy Manipulation Attacks on Deep Q-Networks with Parameter-Space Noise","date":"2018-06-04","arxiv_id":"1806.02190","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-test-for-the-lowest-mean-from","title":"Sequential Test for the Lowest Mean: From Thompson to Murphy Sampling","date":"2018-06-04","arxiv_id":"1806.00973","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-structured-reinforcement","title":"Exploration in Structured Reinforcement Learning","date":"2018-06-03","arxiv_id":"1806.00775","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-via-double","title":"Multi-Agent Reinforcement Learning via Double Averaging Primal-Dual Optimization","date":"2018-06-03","arxiv_id":"1806.00877","repositories_listed":0,"syntology":null},{"url":null,"slug":"daqn-deep-auto-encoder-and-q-network","title":"DAQN: Deep Auto-encoder and Q-Network","date":"2018-06-02","arxiv_id":"1806.00630","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-pepper-expert-iteration-based-chess","title":"Deep Pepper: Expert Iteration based Chess agent in the Reinforcement Learning Setting","date":"2018-06-02","arxiv_id":"1806.00683","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-entropy-for-policy-gradient-with","title":"Efficient Entropy for Policy Gradient with Multidimensional Action Space","date":"2018-06-02","arxiv_id":"1806.00589","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-model-from-observations-for-reward","title":"Internal Model from Observations for Reward Shaping","date":"2018-06-02","arxiv_id":"1806.01267","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-age-of","title":"A Reinforcement Learning Approach to Age of Information in Multi-User Networks","date":"2018-06-01","arxiv_id":"1806.00336","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-a-neural-conversational-agent","title":"Bootstrapping a Neural Conversational Agent with Dialogue Self-Play, Crowdsourcing and On-Line Reinforcement Learning","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/deep-progressive-reinforcement-learning-for","slug":"deep-progressive-reinforcement-learning-for","title":"Deep Progressive Reinforcement Learning for Skeleton-Based Action Recognition","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-upgrade-reinforcement-learning","title":"Environment Upgrade Reinforcement Learning for Non-Differentiable Multi-Stage Pipelines","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"equivalence-between-wasserstein-and-value","title":"Equivalence Between Wasserstein and Value-Aware Loss for Model-based Reinforcement Learning","date":"2018-06-01","arxiv_id":"1806.01265","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-exploration-with-simplified-models-and","title":"Fast Exploration with Simplified Models and Approximately Optimistic Planning in Model Based Reinforcement Learning","date":"2018-06-01","arxiv_id":"1806.00175","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphbit-bitwise-interaction-mining-via-deep","title":"GraphBit: Bitwise Interaction Mining via Deep Reinforcement Learning","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-oracle-complexity-for-stochastic","title":"Improved Sample Complexity for Stochastic Compositional Variance Reduced Gradient","date":"2018-06-01","arxiv_id":"1806.00458","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-aided-reinforcement-learning-for","title":"Inference Aided Reinforcement Learning for Incentive Mechanism Design in Crowdsourcing","date":"2018-06-01","arxiv_id":"1806.00206","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-episodic-memory-into-a","title":"Integrating Episodic Memory into a Reinforcement Learning Agent using Reservoir Sampling","date":"2018-06-01","arxiv_id":"1806.00540","repositories_listed":0,"syntology":null},{"url":null,"slug":"mining-evidences-for-concept-stock","title":"Mining Evidences for Concept Stock Recommendation","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-signals-in-generated-stories","title":"Quality Signals in Generated Stories","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"seednet-automatic-seed-generation-with-deep","title":"SeedNet: Automatic Seed Generation With Deep Reinforcement Learning for Robust Interactive Segmentation","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-reinforcement-learning-algorithms","title":"Evaluating Reinforcement Learning Algorithms in Observational Health Settings","date":"2018-05-31","arxiv_id":"1805.12298","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-prior-over-intent-via-meta-inverse","title":"Learning a Prior over Intent via Meta-Inverse Reinforcement Learning","date":"2018-05-31","arxiv_id":"1805.12573","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-attacks-on-agents-for-long-term","title":"Sequential Attacks on Agents for Long-Term Adversarial Goals","date":"2018-05-31","arxiv_id":"1805.12487","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-learning-of-task-oriented-neural","title":"Adversarial Learning of Task-Oriented Neural Dialog Models","date":"2018-05-30","arxiv_id":"1805.11762","repositories_listed":0,"syntology":null},{"url":null,"slug":"depth-and-nonlinearity-induce-implicit","title":"Depth and nonlinearity induce implicit exploration for RL","date":"2018-05-29","arxiv_id":"1805.11711","repositories_listed":0,"syntology":null},{"url":null,"slug":"observe-and-look-further-achieving-consistent","title":"Observe and Look Further: Achieving Consistent Performance on Atari","date":"2018-05-29","arxiv_id":"1805.11593","repositories_listed":0,"syntology":null},{"url":null,"slug":"truncated-horizon-policy-search-combining","title":"Truncated Horizon Policy Search: Combining Reinforcement Learning & Imitation Learning","date":"2018-05-29","arxiv_id":"1805.11240","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-inverse-control-with-events-a","title":"Variational Inverse Control with Events: A General Framework for Data-Driven Reward Definition","date":"2018-05-29","arxiv_id":"1805.11686","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtuously-safe-reinforcement-learning","title":"Virtuously Safe Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11447","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-clustering-with-deep-q-learning","title":"Hierarchical clustering with deep Q-learning","date":"2018-05-28","arxiv_id":"1805.10900","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-weighted-transfer-of-samples-in","title":"Importance Weighted Transfer of Samples in Reinforcement Learning","date":"2018-05-28","arxiv_id":"1805.10886","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-propagation-networks","title":"Value Propagation Networks","date":"2018-05-28","arxiv_id":"1805.11199","repositories_listed":0,"syntology":null},{"url":null,"slug":"fingerprint-policy-optimisation-for-robust","title":"Fingerprint Policy Optimisation for Robust Reinforcement Learning","date":"2018-05-27","arxiv_id":"1805.10662","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-ice-hockey-for","title":"Deep Reinforcement Learning in Ice Hockey for Context-Aware Player Evaluation","date":"2018-05-26","arxiv_id":"1805.11088","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-policy-learning-through-imitation-and","title":"Fast Policy Learning through Imitation and Reinforcement","date":"2018-05-26","arxiv_id":"1805.10413","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-sliding-window-algorithm-for-markov","title":"A Sliding-Window Algorithm for Markov Decision Processes with Arbitrarily Changing Rewards and Transitions","date":"2018-05-25","arxiv_id":"1805.10066","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-deceptive-reviews-using-generative","title":"Detecting Deceptive Reviews using Generative Adversarial Networks","date":"2018-05-25","arxiv_id":"1805.10364","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-of-lstd-with-random","title":"Finite Sample Analysis of LSTD with Random Projections and Eligibility Traces","date":"2018-05-25","arxiv_id":"1805.10005","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-extractive-summarization-with","title":"Reinforced Extractive Summarization with Question-Focused Rewards","date":"2018-05-25","arxiv_id":"1805.10392","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-allocation-for-a-wireless","title":"Resource Allocation for a Wireless Coexistence Management System Based on Reinforcement Learning","date":"2018-05-24","arxiv_id":"1806.04702","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-blind-spots-in-reinforcement","title":"Discovering Blind Spots in Reinforcement Learning","date":"2018-05-23","arxiv_id":"1805.08966","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyna-planning-using-a-feature-based","title":"Dyna Planning using a Feature Based Generative Model","date":"2018-05-23","arxiv_id":"1805.10129","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-heterogeneous","title":"Reinforcement Learning for Heterogeneous Teams with PALO Bounds","date":"2018-05-23","arxiv_id":"1805.09267","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-simple-exploration-is-sample-efficient","title":"When Simple Exploration is Sample Efficient: Identifying Sufficient Conditions for Random Exploration to Yield PAC RL Algorithms","date":"2018-05-23","arxiv_id":"1805.09045","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-centralized-deep-multi-agent","title":"Scalable Centralized Deep Multi-Agent Reinforcement Learning via Policy Gradients","date":"2018-05-22","arxiv_id":"1805.08776","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-and-method-for-online-inverse","title":"A Framework and Method for Online Inverse Reinforcement Learning","date":"2018-05-21","arxiv_id":"1805.07871","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-family-of-robust-stochastic","title":"A General Family of Robust Stochastic Operators for Reinforcement Learning","date":"2018-05-21","arxiv_id":"1805.08122","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with-1","title":"Hierarchical Reinforcement Learning with Hindsight","date":"2018-05-21","arxiv_id":"1805.08180","repositories_listed":0,"syntology":null},{"url":"/paper/hierarchically-structured-reinforcement","slug":"hierarchically-structured-reinforcement","title":"Hierarchically Structured Reinforcement Learning for Topically Coherent Visual Story Generation","date":"2018-05-21","arxiv_id":"1805.08191","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-safe-policies-with-expert-guidance","title":"Learning Safe Policies with Expert Guidance","date":"2018-05-21","arxiv_id":"1805.08313","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-step-greedy-policies-in-online-and","title":"Multiple-Step Greedy Policies in Online and Approximate Reinforcement Learning","date":"2018-05-21","arxiv_id":"1805.07956","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-real-world-robot-policies-by","title":"Learning Real-World Robot Policies by Dreaming","date":"2018-05-20","arxiv_id":"1805.07813","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-teach-in-cooperative-multiagent","title":"Learning to Teach in Cooperative Multiagent Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07830","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-memory-deep-q-networks","title":"Episodic Memory Deep Q-Networks","date":"2018-05-19","arxiv_id":"1805.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-theorem-proving","title":"Reinforcement Learning of Theorem Proving","date":"2018-05-19","arxiv_id":"1805.07563","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with-deep","title":"Hierarchical Reinforcement Learning with Deep Nested Agents","date":"2018-05-18","arxiv_id":"1805.07008","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-geometric-input-transformation-methods","title":"Two geometric input transformation methods for fast online reinforcement learning with neural nets","date":"2018-05-18","arxiv_id":"1805.07476","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-resource","title":"Deep Reinforcement Learning for Resource Management in Network Slicing","date":"2018-05-17","arxiv_id":"1805.06591","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-rl-for-container-loading","title":"Evolutionary RL for Container Loading","date":"2018-05-17","arxiv_id":"1805.06664","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-expansion-in-text-based-games","title":"Language Expansion In Text-Based Games","date":"2018-05-17","arxiv_id":"1805.07274","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-retinomorphic-event-stream-for-video","title":"Fast Retinomorphic Event Stream for Video Recognition and Reinforcement Learning","date":"2018-05-16","arxiv_id":"1805.06374","repositories_listed":0,"syntology":null},{"url":null,"slug":"follownet-robot-navigation-by-following","title":"FollowNet: Robot Navigation by Following Natural Language Directions with Deep Reinforcement Learning","date":"2018-05-16","arxiv_id":"1805.06150","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimized-computation-offloading-performance","title":"Optimized Computation Offloading Performance in Virtual Edge Computing Systems via Deep Reinforcement Learning","date":"2018-05-16","arxiv_id":"1805.06146","repositories_listed":0,"syntology":null},{"url":null,"slug":"feedback-based-tree-search-for-reinforcement","title":"Feedback-Based Tree Search for Reinforcement Learning","date":"2018-05-15","arxiv_id":"1805.05935","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-signal-sampling-via-reinforcement","title":"Graph Signal Sampling via Reinforcement Learning","date":"2018-05-15","arxiv_id":"1805.05827","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-human-knowledge-in-tabular","title":"Leveraging human knowledge in tabular reinforcement learning: A study of human subjects","date":"2018-05-15","arxiv_id":"1805.05769","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hierarchical-adaptive-forgetting","title":"The Hierarchical Adaptive Forgetting Variational Filter","date":"2018-05-15","arxiv_id":"1805.05703","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-pass-recurrent-neural-networks-a-memory","title":"Low-pass Recurrent Neural Networks - A memory architecture for longer-term correlation discovery","date":"2018-05-13","arxiv_id":"1805.04955","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-rescheduling-knowledge-using","title":"Generating Rescheduling Knowledge using Reinforcement Learning in a Cognitive Architecture","date":"2018-05-12","arxiv_id":"1805.04752","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-autonomous-reinforcement-learning","title":"Towards Autonomous Reinforcement Learning: Automatic Setting of Hyper-parameters using Bayesian Optimization","date":"2018-05-12","arxiv_id":"1805.04748","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hierarchical-reinforcement-learning","title":"Deep Hierarchical Reinforcement Learning Algorithm in Partially Observable Markov Decision Processes","date":"2018-05-11","arxiv_id":"1805.04419","repositories_listed":0,"syntology":null}],"record_sha256":"4125dc7c66d576e8c24c3f252fad0d3d4f063c58530324c97ac3a0300d9989d0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}