{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/119","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":119,"pages_in_order":132,"rows_per_page":100,"rows":[11801,11900],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/118","next":"/task/reinforcement-learning/papers/120","papers":[{"url":null,"slug":"intelligent-traffic-signal-control-using","title":"Using Reinforcement Learning with Partial Vehicle Detection for Intelligent Traffic Signal Control","date":"2018-07-04","arxiv_id":"1807.01628","repositories_listed":0,"syntology":null},{"url":null,"slug":"region-growing-curriculum-generation-for","title":"Region Growing Curriculum Generation for Reinforcement Learning","date":"2018-07-04","arxiv_id":"1807.01425","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-reinforcement-learning-with","title":"Supervised Reinforcement Learning with Recurrent Neural Network for Dynamic Treatment Recommendation","date":"2018-07-04","arxiv_id":"1807.01473","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-with-model-features-in-reinforcement","title":"Transfer with Model Features in Reinforcement Learning","date":"2018-07-04","arxiv_id":"1807.01736","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-level-performance-in-first-person","title":"Human-level performance in first-person multiplayer games with population-based deep reinforcement learning","date":"2018-07-03","arxiv_id":"1807.01281","repositories_listed":0,"syntology":null},{"url":null,"slug":"speeding-up-the-metabolism-in-e-commerce-by","title":"Speeding up the Metabolism in E-commerce by Reinforcement Mechanism Design","date":"2018-07-02","arxiv_id":"1807.00448","repositories_listed":0,"syntology":null},{"url":"/paper/a-language-model-based-evaluator-for-sentence","slug":"a-language-model-based-evaluator-for-sentence","title":"A Language Model based Evaluator for Sentence Compression","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-neural-network-for","title":"A Reinforcement Learning Neural Network for Robotic Manipulator Control","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-one-step-greedy-approach-in","title":"Beyond the One-Step Greedy Approach in Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-winning-and-losing-modeling-human","title":"Beyond Winning and Losing: Modeling Human Motivations and Behaviors Using Inverse Reinforcement Learning","date":"2018-07-01","arxiv_id":"1807.00366","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupling-gradient-like-learning-rules-from","title":"Decoupling Gradient-Like Learning Rules from Representations","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-nlp","title":"Deep Reinforcement Learning for NLP","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"feudal-dialogue-management-with-jointly","title":"Feudal Dialogue Management with Jointly Learned Feature Extractors","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-regret-bounds-for-thompson-sampling","title":"Improved Regret Bounds for Thompson Sampling in Linear Quadratic Control Problems","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-hierarchical-structures-on-the-fly","title":"Learning Hierarchical Structures On-The-Fly with a Recurrent-Recursive Model for Sequences","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-act-in-decentralized-partially","title":"Learning to Act in Decentralized Partially Observable MDPs","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-coordinate-with-coordination","title":"Learning to Coordinate with Coordination Graphs in Repeated Single-Stage Multi-Agent Decision Problems","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-explore-via-meta-policy-gradient","title":"Learning to Explore via Meta-Policy Gradient","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-match-agent-curricula-for-reinforcement","title":"Mix & Match - Agent Curricula for Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-user-simulation-for-corpus-based","title":"Neural User Simulation for Corpus-based Policy Optimisation of Spoken Dialogue Systems","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-and-value-transfer-in-lifelong","title":"Policy and Value Transfer in Lifelong Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-with-demonstrations","title":"Policy Optimization with Demonstrations","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-bilinear-pi-learning-using-state-and","title":"Scalable Bilinear Pi Learning Using State and Action Features","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"spotlight-optimizing-device-placement-for","title":"Spotlight: Optimizing Device Placement for Training Deep Neural Networks","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"state-abstractions-for-lifelong-reinforcement","title":"State Abstractions for Lifelong Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-importance-of-recommender-and-feedback","title":"The Importance of Recommender and Feedback Features in a Pronunciation Learning Aid","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-mixed-optimization-for-reinforcement","title":"Towards Mixed Optimization for Reinforcement Learning with Program Synthesis","date":"2018-07-01","arxiv_id":"1807.00403","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-simplifying-one-shot","title":"Understanding and Simplifying One-Shot Architecture Search","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with","title":"Hierarchical Reinforcement Learning with Abductive Planning","date":"2018-06-28","arxiv_id":"1806.10792","repositories_listed":0,"syntology":null},{"url":null,"slug":"monas-multi-objective-neural-architecture","title":"MONAS: Multi-Objective Neural Architecture Search using Reinforcement Learning","date":"2018-06-27","arxiv_id":"1806.10332","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-exploration-strategy-for-self","title":"Adversarial Active Exploration for Inverse Dynamics Model Learning","date":"2018-06-26","arxiv_id":"1806.10019","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-generative-models-with-learnable","title":"Deep Generative Models with Learnable Knowledge Constraints","date":"2018-06-26","arxiv_id":"1806.09764","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-existing-social-conventions-in","title":"Learning Existing Social Conventions via Observationally Augmented Self-Play","date":"2018-06-26","arxiv_id":"1806.10071","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-inverse-reinforcement-learning-1","title":"Multi-agent Inverse Reinforcement Learning for Certain General-sum Stochastic Games","date":"2018-06-26","arxiv_id":"1806.09795","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-an-overview-1","title":"Deep Reinforcement Learning: An Overview","date":"2018-06-23","arxiv_id":"1806.08894","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-interactive-subgoal-supervision-for","title":"Human-Interactive Subgoal Supervision for Efficient Inverse Reinforcement Learning","date":"2018-06-22","arxiv_id":"1806.08479","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-ask-knowledge-acquisition-via-20","title":"Learning-to-Ask: Knowledge Acquisition via 20 Questions","date":"2018-06-22","arxiv_id":"1806.08554","repositories_listed":0,"syntology":null},{"url":null,"slug":"many-goals-reinforcement-learning","title":"Many-Goals Reinforcement Learning","date":"2018-06-22","arxiv_id":"1806.09605","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-approach-for-resource-scheduling-with","title":"A New Approach for Resource Scheduling with Deep Reinforcement Learning","date":"2018-06-21","arxiv_id":"1806.08122","repositories_listed":0,"syntology":null},{"url":null,"slug":"expanding-the-active-inference-landscape-more","title":"Expanding the Active Inference Landscape: More Intrinsic Motivations in the Perception-Action Loop","date":"2018-06-21","arxiv_id":"1806.08083","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dissection-of-overfitting-and","title":"A Dissection of Overfitting and Generalization in Continuous Reinforcement Learning","date":"2018-06-20","arxiv_id":"1806.07937","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-neural-parsers-with-deterministic","title":"Learning Neural Parsers with Deterministic Differentiable Imitation Learning","date":"2018-06-20","arxiv_id":"1806.07822","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-augmented-neural","title":"Reinforcement Learning using Augmented Neural Networks","date":"2018-06-20","arxiv_id":"1806.07692","repositories_listed":0,"syntology":null},{"url":null,"slug":"skilled-experience-catalogue-a-skill","title":"Skilled Experience Catalogue: A Skill-Balancing Mechanism for Non-Player Characters using Reinforcement Learning","date":"2018-06-20","arxiv_id":"1806.07637","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-inverse-reinforcement-learning","title":"A Survey of Inverse Reinforcement Learning: Challenges, Methods and Progress","date":"2018-06-18","arxiv_id":"1806.06877","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-strategy-for-implementing-curiosity","title":"A unified strategy for implementing curiosity and empowerment driven reinforcement learning","date":"2018-06-18","arxiv_id":"1806.06505","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-outside-the-viability-kernel","title":"Learning from Outside the Viability Kernel: Why we Should Build Robots that can Fall with Grace","date":"2018-06-18","arxiv_id":"1806.06569","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-policy-representations-in-multiagent","title":"Learning Policy Representations in Multiagent Systems","date":"2018-06-17","arxiv_id":"1806.06464","repositories_listed":0,"syntology":null},{"url":null,"slug":"handling-cold-start-collaborative-filtering","title":"Handling Cold-Start Collaborative Filtering with Reinforcement Learning","date":"2018-06-16","arxiv_id":"1806.06192","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-online-prediction-algorithm-for","title":"An Online Prediction Algorithm for Reinforcement Learning with Linear Function Approximation using Cross Entropy Method","date":"2018-06-15","arxiv_id":"1806.06720","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-width-based-planning-with-compact","title":"Improving width-based planning with compact policies","date":"2018-06-15","arxiv_id":"1806.05898","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-policy-and-reward-reinforcement","title":"Multi-Level Policy and Reward Reinforcement Learning for Image Captioning","date":"2018-06-15","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-shooting-for-bots-in-first-person","title":"Adaptive Shooting for Bots in First Person Shooter Games Using Reinforcement Learning","date":"2018-06-14","arxiv_id":"1806.05554","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dynamic-urban","title":"Deep Reinforcement Learning for Dynamic Urban Transportation Problems","date":"2018-06-14","arxiv_id":"1806.05310","repositories_listed":0,"syntology":null},{"url":null,"slug":"qualitative-measurements-of-policy","title":"Qualitative Measurements of Policy Discrepancy for Return-Based Deep Q-Network","date":"2018-06-14","arxiv_id":"1806.06953","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-formation-of-the-structure-of","title":"Automatic formation of the structure of abstract machines in hierarchical reinforcement learning with state clustering","date":"2018-06-13","arxiv_id":"1806.05292","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-shoot-in-first-person-shooter","title":"Learning to Shoot in First Person Shooter Games by Stabilizing Actions and Clustering Rewards for Reinforcement Learning","date":"2018-06-13","arxiv_id":"1806.05117","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-function-valued","title":"Reinforcement Learning with Function-Valued Action Spaces for Partial Differential Equation Control","date":"2018-06-13","arxiv_id":"1806.06931","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-imitation-learning-with","title":"Accelerating Imitation Learning with Predictive Models","date":"2018-06-12","arxiv_id":"1806.04642","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-regression-performance-with","title":"Improving Regression Performance with Distributional Losses","date":"2018-06-12","arxiv_id":"1806.04613","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-transferable-active-learning","title":"Meta-Learning Transferable Active Learning Policies by Deep Reinforcement Learning","date":"2018-06-12","arxiv_id":"1806.04798","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-with-1","title":"Multi-Agent Deep Reinforcement Learning with Human Strategies","date":"2018-06-12","arxiv_id":"1806.04562","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-meta-learning-for-reinforcement","title":"Unsupervised Meta-Learning for Reinforcement Learning","date":"2018-06-12","arxiv_id":"1806.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-generalized-bellman-update-for","title":"An Efficient, Generalized Bellman Update For Cooperative Inverse Reinforcement Learning","date":"2018-06-11","arxiv_id":"1806.03820","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-policy-reuse","title":"Context-Aware Policy Reuse","date":"2018-06-11","arxiv_id":"1806.03793","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-curiosity-loops-in-social-environments","title":"Deep Curiosity Loops in Social Environments","date":"2018-06-10","arxiv_id":"1806.03645","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-policy-for-reinforcement-learning","title":"Implicit Policy for Reinforcement Learning","date":"2018-06-10","arxiv_id":"1806.06798","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-view-planning-with-multi-scale-deep","title":"Automatic View Planning with Multi-scale Deep Reinforcement Learning Agents","date":"2018-06-08","arxiv_id":"1806.03228","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-time-value-function-approximation","title":"Continuous-time Value Function Approximation in Reproducing Kernel Hilbert Spaces","date":"2018-06-08","arxiv_id":"1806.02985","repositories_listed":0,"syntology":null},{"url":null,"slug":"fidelity-based-probabilistic-q-learning-for","title":"Fidelity-based Probabilistic Q-learning for Control of Quantum Systems","date":"2018-06-08","arxiv_id":"1806.03145","repositories_listed":0,"syntology":null},{"url":null,"slug":"program-synthesis-through-reinforcement","title":"Program Synthesis Through Reinforcement Learning Guided Tree Search","date":"2018-06-08","arxiv_id":"1806.02932","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-consistent-trajectory-autoencoder","title":"Self-Consistent Trajectory Autoencoder: Hierarchical Reinforcement Learning with Trajectory Embeddings","date":"2018-06-07","arxiv_id":"1806.02813","repositories_listed":0,"syntology":null},{"url":null,"slug":"simplifying-reward-design-through-divide-and","title":"Simplifying Reward Design through Divide-and-Conquer","date":"2018-06-07","arxiv_id":"1806.02501","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-time-analysis-of-temporal-difference","title":"A Finite Time Analysis of Temporal Difference Learning With Linear Function Approximation","date":"2018-06-06","arxiv_id":"1806.02450","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-by-the-baldwin-effect","title":"Meta-Learning by the Baldwin Effect","date":"2018-06-06","arxiv_id":"1806.07917","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-and-removing-exogenous-state","title":"Discovering and Removing Exogenous State Variables and Rewards for Reinforcement Learning","date":"2018-06-05","arxiv_id":"1806.01584","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixmatch-agent-curricula-for-reinforcement","title":"Mix&Match - Agent Curricula for Reinforcement Learning","date":"2018-06-05","arxiv_id":"1806.01780","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-effect-of-planning-shape-on-dyna-style","title":"The Effect of Planning Shape on Dyna-style Planning in High-dimensional State Spaces","date":"2018-06-05","arxiv_id":"1806.01825","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-framework","title":"Adversarial Reinforcement Learning Framework for Benchmarking Collision Avoidance Mechanisms in Autonomous Vehicles","date":"2018-06-04","arxiv_id":"1806.01368","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-and-avoiding-side-effects-using","title":"Penalizing side effects using stepwise relative reachability","date":"2018-06-04","arxiv_id":"1806.01186","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigation-of-policy-manipulation-attacks-on","title":"Mitigation of Policy Manipulation Attacks on Deep Q-Networks with Parameter-Space Noise","date":"2018-06-04","arxiv_id":"1806.02190","repositories_listed":0,"syntology":null},{"url":null,"slug":"relational-inductive-bias-for-physical","title":"Relational inductive bias for physical construction in humans and machines","date":"2018-06-04","arxiv_id":"1806.01203","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-test-for-the-lowest-mean-from","title":"Sequential Test for the Lowest Mean: From Thompson to Murphy Sampling","date":"2018-06-04","arxiv_id":"1806.00973","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-advanced-dialogue-managers-for-goal","title":"Building Advanced Dialogue Managers for Goal-Oriented Dialogue Systems","date":"2018-06-03","arxiv_id":"1806.00780","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-structured-reinforcement","title":"Exploration in Structured Reinforcement Learning","date":"2018-06-03","arxiv_id":"1806.00775","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-via-double","title":"Multi-Agent Reinforcement Learning via Double Averaging Primal-Dual Optimization","date":"2018-06-03","arxiv_id":"1806.00877","repositories_listed":0,"syntology":null},{"url":null,"slug":"daqn-deep-auto-encoder-and-q-network","title":"DAQN: Deep Auto-encoder and Q-Network","date":"2018-06-02","arxiv_id":"1806.00630","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-pepper-expert-iteration-based-chess","title":"Deep Pepper: Expert Iteration based Chess agent in the Reinforcement Learning Setting","date":"2018-06-02","arxiv_id":"1806.00683","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-entropy-for-policy-gradient-with","title":"Efficient Entropy for Policy Gradient with Multidimensional Action Space","date":"2018-06-02","arxiv_id":"1806.00589","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-model-from-observations-for-reward","title":"Internal Model from Observations for Reward Shaping","date":"2018-06-02","arxiv_id":"1806.01267","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-age-of","title":"A Reinforcement Learning Approach to Age of Information in Multi-User Networks","date":"2018-06-01","arxiv_id":"1806.00336","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-a-neural-conversational-agent","title":"Bootstrapping a Neural Conversational Agent with Dialogue Self-Play, Crowdsourcing and On-Line Reinforcement Learning","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-curiosity-search-intra-life-exploration","title":"Deep Curiosity Search: Intra-Life Exploration Can Improve Performance on Challenging Deep Reinforcement Learning Problems","date":"2018-06-01","arxiv_id":"1806.00553","repositories_listed":0,"syntology":null},{"url":"/paper/deep-progressive-reinforcement-learning-for","slug":"deep-progressive-reinforcement-learning-for","title":"Deep Progressive Reinforcement Learning for Skeleton-Based Action Recognition","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"egocentric-activity-recognition-on-a-budget","title":"Egocentric Activity Recognition on a Budget","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-learning-of-task-oriented-dialogs","title":"End-to-End Learning of Task-Oriented Dialogs","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-upgrade-reinforcement-learning","title":"Environment Upgrade Reinforcement Learning for Non-Differentiable Multi-Stage Pipelines","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"equivalence-between-wasserstein-and-value","title":"Equivalence Between Wasserstein and Value-Aware Loss for Model-based Reinforcement Learning","date":"2018-06-01","arxiv_id":"1806.01265","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-exploration-with-simplified-models-and","title":"Fast Exploration with Simplified Models and Approximately Optimistic Planning in Model Based Reinforcement Learning","date":"2018-06-01","arxiv_id":"1806.00175","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphbit-bitwise-interaction-mining-via-deep","title":"GraphBit: Bitwise Interaction Mining via Deep Reinforcement Learning","date":"2018-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"aebaff74d371302024857b9b5a57f70f8b163605e9f9b5657f925d9a2553110b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}