{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/140","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":140,"pages_in_order":152,"rows_per_page":100,"rows":[13901,14000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/139","next":"/task/reinforcement-learning-1/papers/141","papers":[{"url":null,"slug":"adaptive-multi-pass-decoder-for-neural","title":"Adaptive Multi-pass Decoder for Neural Machine Translation","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-essay-scoring-incorporating-rating","title":"Automatic Essay Scoring Incorporating Rating Schema via Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-poetry-generation-with-mutual","title":"Automatic Poetry Generation with Mutual Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-sub-domain-modeling-for-dialogue","title":"Autonomous Sub-domain Modeling for Dialogue Policy with Hierarchical Deep Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-based-on-reward","title":"Curriculum Learning Based on Reward Sparseness for Deep Reinforcement Learning of Task Completion Dialogue Management","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"logician-and-orator-learning-from-the-duality","title":"Logician and Orator: Learning from the Duality between Language and Knowledge in Open Domain","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-improves-simultaneous-neural","title":"Prediction Improves Simultaneous Neural Machine Translation","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"smartchoices-hybridizing-programming-and","title":"SmartChoices: Hybridizing Programming and Machine Learning","date":"2018-10-01","arxiv_id":"1810.00619","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-transfer-reinforcement-learning-with","title":"Bayesian Transfer Reinforcement Learning with Prior Knowledge Rules","date":"2018-09-30","arxiv_id":"1810.00468","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-goal-inference-for-visuomotor","title":"Few-Shot Goal Inference for Visuomotor Learning and Planning","date":"2018-09-30","arxiv_id":"1810.00482","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-r","title":"Reinforcement Learning in R","date":"2018-09-29","arxiv_id":"1810.00240","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-optimization-of-f-measure-for","title":"Direct optimization of F-measure for retrieval-based personal question answering","date":"2018-09-28","arxiv_id":"1810.00679","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-representation-and-reasoning-with","title":"Robot Representation and Reasoning with Knowledge from Reinforcement Learning","date":"2018-09-28","arxiv_id":"1809.11074","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-better-baseline-for-second-order-gradient","title":"A Better Baseline for Second Order Gradient Estimation in Stochastic Computation Graphs","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-convergent-variant-of-the-boltzmann-softmax","title":"A Convergent Variant of the Boltzmann Softmax Operator in Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-value-iteration-via-anderson","title":"Accelerated Value Iteration via Anderson Mixing","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-multiagent-reinforcement","title":"COLLABORATIVE MULTIAGENT REINFORCEMENT LEARNING IN HOMOGENEOUS SWARMS","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"constraining-action-sequences-with-formal","title":"Constraining Action Sequences with Formal Languages for Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"convergent-reinforcement-learning-with","title":"Convergent Reinforcement Learning with Function Approximation: A Bilevel Optimization Perspective","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"countering-language-drift-via-grounding","title":"Countering Language Drift via Grounding","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-adversarial-forward-model","title":"DEEP ADVERSARIAL FORWARD MODEL","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-of-universal","title":"Deep Reinforcement Learning of Universal Policies with Diverse Environment Summaries","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"definition-and-evaluation-of-model-free","title":"Definition and evaluation of model-free coordination of electrical vehicle charging with reinforcement learning","date":"2018-09-27","arxiv_id":"1809.10679","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilled-agent-dqn-for-provable-adversarial","title":"Distilled Agent DQN for Provable Adversarial Robustness","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-via-distribution-and","title":"DOMAIN ADAPTATION VIA DISTRIBUTION AND REPRESENTATION MATCHING: A CASE STUDY ON TRAINING DATA SELECTION VIA REINFORCEMENT LEARNING","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-pricing-on-e-commerce-platform-with-1","title":"Dynamic Pricing on E-commerce Platform with Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-environmental-variation-to-improve","title":"Exploiting Environmental Variation to Improve Policy Robustness in Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-by-uncertainty-in-reward-space","title":"Exploration by Uncertainty in Reward Space","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"expressiveness-in-deep-reinforcement-learning","title":"Expressiveness in Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-exploration-in-deep-reinforcement","title":"Guided Exploration in Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-policies-using-inverse-rewards-for","title":"Hybrid Policies Using Inverse Rewards for Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-hierarchical-reinforcement","title":"Incremental Hierarchical Reinforcement Learning with Multitask LMDPs","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-parallel-exploration-for","title":"Interactive Parallel Exploration for Reinforcement Learning in Continuous Action Spaces","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-physics-priors-for-deep","title":"Learning Physics Priors for Deep Reinforcement Learing","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-coordinate-multiple-reinforcement","title":"Learning to Coordinate Multiple Reinforcement Learning Agents for Diverse Query Reformulation","date":"2018-09-27","arxiv_id":"1809.10658","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimicking-actions-is-a-good-strategy-for","title":"Mimicking actions is a good strategy for beginners: Fast Reinforcement Learning with Expert Action Sequences","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-generalization-in-capacity-limited","title":"Policy Generalization In Capacity-Limited Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"shrinkage-based-bias-variance-trade-off-for","title":"Shrinkage-based Bias-Variance Trade-off for Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-options-an-option-discovery-1","title":"Successor Options : An Option Discovery Algorithm for Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-wisdom-of-the-crowd-reliable-deep","title":"The wisdom of the crowd: reliable deep reinforcement learning through ensembles of Q-functions","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-more-theoretically-grounded-particle","title":"Towards More Theoretically-Grounded Particle Optimization Sampling for Deep Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-value-or-policy-a-value-centric","title":"Transfer Value or Policy? A Value-centric Framework Towards Transferrable Continuous Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-exploration-with-deep-model","title":"Unsupervised Exploration with Deep Model-Based Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"what-would-pi-do-imitation-learning-via-off","title":"What Would pi* Do?: Imitation Learning via Off-Policy Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"where-off-policy-deep-reinforcement-learning","title":"Where Off-Policy Deep Reinforcement Learning Fails","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alphaseq-sequence-discovery-with-deep","title":"AlphaSeq: Sequence Discovery with Deep Reinforcement Learning","date":"2018-09-26","arxiv_id":"1810.01218","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-navigation-behaviors-end-to-end-with","title":"Learning Navigation Behaviors End-to-End with AutoRL","date":"2018-09-26","arxiv_id":"1809.10124","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-through-probing-a-decentralized","title":"Learning through Probing: a decentralized reinforcement learning architecture for social dilemmas","date":"2018-09-26","arxiv_id":"1809.10007","repositories_listed":0,"syntology":null},{"url":null,"slug":"omega-regular-objectives-in-model-free","title":"Omega-Regular Objectives in Model-Free Reinforcement Learning","date":"2018-09-26","arxiv_id":"1810.00950","repositories_listed":0,"syntology":null},{"url":null,"slug":"anderson-acceleration-for-reinforcement","title":"Anderson Acceleration for Reinforcement Learning","date":"2018-09-25","arxiv_id":"1809.09501","repositories_listed":0,"syntology":null},{"url":null,"slug":"floyd-warshall-reinforcement-learning","title":"Floyd-Warshall Reinforcement Learning: Learning from Past Experiences to Reach New Goals","date":"2018-09-25","arxiv_id":"1809.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-deep-multiagent-reinforcement","title":"Hierarchical Deep Multiagent Reinforcement Learning with Temporal Abstraction","date":"2018-09-25","arxiv_id":"1809.09332","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-precision-policy-distillation-with","title":"Low Precision Policy Distillation with Application to Low-Power, Real-time Sensation-Cognition-Action Loop with Neuromorphic Computing","date":"2018-09-25","arxiv_id":"1809.09260","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-computing-with-reinforcement","title":"Resilient Computing with Reinforcement Learning on a Dynamical System: Case Study in Sorting","date":"2018-09-25","arxiv_id":"1809.09261","repositories_listed":0,"syntology":null},{"url":null,"slug":"epirl-a-reinforcement-learning-agent-to","title":"EpiRL: A Reinforcement Learning Agent to Facilitate Epistasis Detection","date":"2018-09-24","arxiv_id":"1809.09143","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-education-at-scale","title":"Personalized Education at Scale","date":"2018-09-24","arxiv_id":"1809.10025","repositories_listed":0,"syntology":null},{"url":null,"slug":"sdn-flow-entry-management-using-reinforcement","title":"SDN Flow Entry Management Using Reinforcement Learning","date":"2018-09-24","arxiv_id":"1809.09003","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learning-framework-for-high-precision","title":"A Learning Framework for High Precision Industrial Assembly","date":"2018-09-23","arxiv_id":"1809.08548","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reinforcement-learning-for-full-length","title":"On Reinforcement Learning for Full-length Game of StarCraft","date":"2018-09-23","arxiv_id":"1809.09095","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-multi-model-fitting-by-deep","title":"Geometric Multi-Model Fitting by Deep Reinforcement Learning","date":"2018-09-22","arxiv_id":"1809.08397","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-of-the-gtd-policy","title":"Finite Sample Analysis of the GTD Policy Evaluation Algorithms in Markov Setting","date":"2018-09-21","arxiv_id":"1809.08926","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-multi-objective-reinforcement","title":"Interpretable Multi-Objective Reinforcement Learning through Policy Orchestration","date":"2018-09-21","arxiv_id":"1809.08343","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-transfer-q-learning-and-its","title":"Target Transfer Q-Learning and Its Convergence Analysis","date":"2018-09-21","arxiv_id":"1809.08923","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligentcrowd-mobile-crowdsensing-via","title":"IntelligentCrowd: Mobile Crowdsensing via Multi-Agent Reinforcement Learning","date":"2018-09-20","arxiv_id":"1809.07830","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-sim-to-real-transfer-with-modular","title":"Sim-to-Real Transfer of Robot Learning with Variable Length Inputs","date":"2018-09-20","arxiv_id":"1809.07480","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-reinforcement-learning-with","title":"Interpretable Reinforcement Learning with Ensemble Methods","date":"2018-09-19","arxiv_id":"1809.06995","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosocial-or-selfish-agents-with-different","title":"Prosocial or Selfish? Agents with different behaviors for Contract Negotiation using Reinforcement Learning","date":"2018-09-19","arxiv_id":"1809.07066","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiobjective-reinforcement-learning-for","title":"Multiobjective Reinforcement Learning for Reconfigurable Adaptive Optimal Control of Manufacturing Processes","date":"2018-09-18","arxiv_id":"1809.06750","repositories_listed":0,"syntology":null},{"url":null,"slug":"scc-rfmq-learning-in-cooperative-markov-games","title":"SCC-rFMQ Learning in Cooperative Markov Games with Continuous Actions","date":"2018-09-18","arxiv_id":"1809.06625","repositories_listed":0,"syntology":null},{"url":null,"slug":"switching-isotropic-and-directional","title":"Switching Isotropic and Directional Exploration with Parameter Space Noise in Deep Reinforcement Learning","date":"2018-09-18","arxiv_id":"1809.06570","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-imitation-via-variational-inverse","title":"Adversarial Imitation via Variational Inverse Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06404","repositories_listed":0,"syntology":null},{"url":null,"slug":"automata-guided-reinforcement-learning-with","title":"Automata Guided Reinforcement Learning With Demonstrations","date":"2018-09-17","arxiv_id":"1809.06305","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-goal-masking-for-continuous-deep","title":"Curriculum goal masking for continuous deep reinforcement learning","date":"2018-09-17","arxiv_id":"1809.06146","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-collaborate-multi-scenario","title":"Learning to Collaborate: Multi-Scenario Ranking via Multi-Agent Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06260","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-sensitive-deep-reinforcement-learning","title":"Object-sensitive Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"improvements-on-hindsight-learning","title":"Improvements on Hindsight Learning","date":"2018-09-16","arxiv_id":"1809.06719","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-for","title":"Adversarial Reinforcement Learning for Observer Design in Autonomous Systems under Cyber Attacks","date":"2018-09-15","arxiv_id":"1809.06784","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-tuning-distributed-stream-processing","title":"Auto-tuning Distributed Stream Processing Systems using Reinforcement Learning","date":"2018-09-14","arxiv_id":"1809.05495","repositories_listed":0,"syntology":null},{"url":null,"slug":"macquarie-university-at-bioasq-6b-deep-1","title":"Macquarie University at BioASQ 6b: Deep learning and deep reinforcement learning for query-based multi-document summarisation","date":"2018-09-14","arxiv_id":"1809.05283","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-diagnostics-for-deep-reinforcement","title":"Visual Diagnostics for Deep Reinforcement Learning Policy Development","date":"2018-09-14","arxiv_id":"1809.06781","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordination-driven-learning-in-multi-agent","title":"Coordination-driven learning in multi-agent problem spaces","date":"2018-09-13","arxiv_id":"1809.04918","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-captioning-based-on-deep-reinforcement","title":"Image Captioning based on Deep Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.04835","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-personalized-and-flexible-playlist","title":"Automatic, Personalized, and Flexible Playlist Generation using Reinforcement Learning","date":"2018-09-12","arxiv_id":"1809.04214","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-topology-based","title":"Reinforcement Learning in Topology-based Representation for Human Body Movement with Whole Arm Manipulation","date":"2018-09-12","arxiv_id":"1809.04322","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-reinforcement-learning-method","title":"A Multi-Agent Reinforcement Learning Method for Impression Allocation in Online Display Advertising","date":"2018-09-10","arxiv_id":"1809.03152","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-behavior-generation-for-autonomous","title":"Adaptive Behavior Generation for Autonomous Driving using Deep Reinforcement Learning with Compact Semantic States","date":"2018-09-10","arxiv_id":"1809.03214","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-generate-structured-queries-from","title":"Learning to Generate Structured Queries from Natural Language with Indirect Supervision","date":"2018-09-10","arxiv_id":"1809.03195","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-one-shot-learning-for-rare-word","title":"Towards one-shot learning for rare-word translation with external experts","date":"2018-09-10","arxiv_id":"1809.03182","repositories_listed":0,"syntology":null},{"url":null,"slug":"vpe-variational-policy-embedding-for-transfer","title":"VPE: Variational Policy Embedding for Transfer Reinforcement Learning","date":"2018-09-10","arxiv_id":"1809.03548","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-prediction-of-interactive","title":"Probabilistic Prediction of Interactive Driving Behavior via Hierarchical Inverse Reinforcement Learning","date":"2018-09-09","arxiv_id":"1809.02926","repositories_listed":0,"syntology":null},{"url":null,"slug":"ans-adaptive-network-scaling-for-deep","title":"ANS: Adaptive Network Scaling for Deep Rectifier Reinforcement Learning Models","date":"2018-09-06","arxiv_id":"1809.02112","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-combine-tree-search-methods-in","title":"How to Combine Tree-Search Methods in Reinforcement Learning","date":"2018-09-06","arxiv_id":"1809.01843","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-what-not-to-learn-action-elimination","title":"Learn What Not to Learn: Action Elimination with Deep Reinforcement Learning","date":"2018-09-06","arxiv_id":"1809.02121","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-regularization-for-deep","title":"Model-Based Regularization for Deep Reinforcement Learning with Transcoder Networks","date":"2018-09-06","arxiv_id":"1809.01906","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-world-models-facilitate-policy","title":"Recurrent World Models Facilitate Policy Evolution","date":"2018-09-04","arxiv_id":"1809.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-deep-reinforcement-learning-with","title":"Transferring Deep Reinforcement Learning with Adversarial Objective and Augmentation","date":"2018-09-04","arxiv_id":"1809.00770","repositories_listed":0,"syntology":null},{"url":null,"slug":"flatland-a-lightweight-first-person-2-d","title":"Flatland: a Lightweight First-Person 2-D Environment for Reinforcement Learning","date":"2018-09-03","arxiv_id":"1809.00510","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-exploration-for-deep-reinforcement","title":"Effective Exploration for Deep Reinforcement Learning via Bootstrapped Q-Ensembles under Tsallis Entropy Regularization","date":"2018-09-02","arxiv_id":"1809.00403","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-person-search-using-deep","title":"Natural Language Person Search Using Deep Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00365","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contextual-bandit-based-approach-for","title":"A Contextual-bandit-based Approach for Informed Decision-making in Clinical Trials","date":"2018-09-01","arxiv_id":"1809.00258","repositories_listed":0,"syntology":null}],"record_sha256":"a3c25803f9471dd06c03bf855e704763d32638c6281257bdd893b99b93e058d2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}