{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/41","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":41,"pages_in_order":132,"rows_per_page":100,"rows":[4001,4100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/40","next":"/task/reinforcement-learning/papers/42","papers":[{"url":"/paper/classification-with-costly-features-using","slug":"classification-with-costly-features-using","title":"Classification with Costly Features using Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07364","repositories_listed":1,"syntology":null},{"url":"/paper/implementing-the-deep-q-network","slug":"implementing-the-deep-q-network","title":"Implementing the Deep Q-Network","date":"2017-11-20","arxiv_id":"1711.07478","repositories_listed":1,"syntology":null},{"url":"/paper/is-prioritized-sweeping-the-better-episodic","slug":"is-prioritized-sweeping-the-better-episodic","title":"Is prioritized sweeping the better episodic control?","date":"2017-11-20","arxiv_id":"1711.06677","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-a-machine-to-read-maps-with-deep","slug":"teaching-a-machine-to-read-maps-with-deep","title":"Teaching a Machine to Read Maps with Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07479","repositories_listed":1,"syntology":null},{"url":"/paper/leave-no-trace-learning-to-reset-for-safe-and","slug":"leave-no-trace-learning-to-reset-for-safe-and","title":"Leave no Trace: Learning to Reset for Safe and Autonomous Reinforcement Learning","date":"2017-11-18","arxiv_id":"1711.06782","repositories_listed":1,"syntology":null},{"url":"/paper/run-skeleton-run-skeletal-model-in-a-physics","slug":"run-skeleton-run-skeletal-model-in-a-physics","title":"Run, skeleton, run: skeletal model in a physics-based simulation","date":"2017-11-18","arxiv_id":"1711.06922","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-policy-gradients","slug":"hindsight-policy-gradients","title":"Hindsight policy gradients","date":"2017-11-16","arxiv_id":"1711.06006","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hindsight-policy-gradients#ran","syntology_url":"https://syntology.ai/paper/1711.06006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.06006"}},"official":null}},{"url":"/paper/classical-structured-prediction-losses-for","slug":"classical-structured-prediction-losses-for","title":"Classical Structured Prediction Losses for Sequence to Sequence Learning","date":"2017-11-14","arxiv_id":"1711.04956","repositories_listed":1,"syntology":null},{"url":"/paper/carla-an-open-urban-driving-simulator","slug":"carla-an-open-urban-driving-simulator","title":"CARLA: An Open Urban Driving Simulator","date":"2017-11-10","arxiv_id":"1711.03938","repositories_listed":1,"syntology":null},{"url":"/paper/towards-the-use-of-deep-reinforcement","slug":"towards-the-use-of-deep-reinforcement","title":"Towards the Use of Deep Reinforcement Learning with Global Policy For Query-based Extractive Summarisation","date":"2017-11-10","arxiv_id":"1711.03859","repositories_listed":1,"syntology":null},{"url":"/paper/latentpoison-adversarial-attacks-on-the","slug":"latentpoison-adversarial-attacks-on-the","title":"LatentPoison - Adversarial Attacks On The Latent Space","date":"2017-11-08","arxiv_id":"1711.02879","repositories_listed":1,"syntology":null},{"url":"/paper/can-deep-reinforcement-learning-solve-erdos","slug":"can-deep-reinforcement-learning-solve-erdos","title":"Can Deep Reinforcement Learning Solve Erdos-Selfridge-Spencer Games?","date":"2017-11-07","arxiv_id":"1711.02301","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-game-theoretic-approach-to","slug":"a-unified-game-theoretic-approach-to","title":"A Unified Game-Theoretic Approach to Multiagent Reinforcement Learning","date":"2017-11-02","arxiv_id":"1711.00832","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-coordination-of-working-memory-and","slug":"adaptive-coordination-of-working-memory-and","title":"Adaptive coordination of working-memory and reinforcement learning in non-human primates performing a trial-and-error problem solving task","date":"2017-11-02","arxiv_id":"1711.00698","repositories_listed":1,"syntology":null},{"url":"/paper/learning-with-latent-language","slug":"learning-with-latent-language","title":"Learning with Latent Language","date":"2017-11-01","arxiv_id":"1711.00482","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-with-latent-language#ran","syntology_url":"https://syntology.ai/paper/1711.00482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.00482"}},"official":{"repos":["jacobandreas/l3"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regret-minimization-for-partially-observable","slug":"regret-minimization-for-partially-observable","title":"Regret Minimization for Partially Observable Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11424","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regret-minimization-for-partially-observable#ran","syntology_url":"https://syntology.ai/paper/1710.11424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11424"}},"official":null}},{"url":"/paper/treeqn-and-atreec-differentiable-tree","slug":"treeqn-and-atreec-differentiable-tree","title":"TreeQN and ATreeC: Differentiable Tree-Structured Models for Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11417","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/treeqn-and-atreec-differentiable-tree#ran","syntology_url":"https://syntology.ai/paper/1710.11417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11417"}},"official":{"repos":["oxwhirl/treeqn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/eigenoption-discovery-through-the-deep","slug":"eigenoption-discovery-through-the-deep","title":"Eigenoption Discovery through the Deep Successor Representation","date":"2017-10-30","arxiv_id":"1710.11089","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eigenoption-discovery-through-the-deep#ran","syntology_url":"https://syntology.ai/paper/1710.11089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11089"}},"official":null}},{"url":"/paper/predicting-head-movement-in-panoramic-video-a","slug":"predicting-head-movement-in-panoramic-video-a","title":"Predicting Head Movement in Panoramic Video: A Deep Reinforcement Learning Approach","date":"2017-10-30","arxiv_id":"1710.10755","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-tower-network-a-novel-deep","slug":"generalization-tower-network-a-novel-deep","title":"Generalization Tower Network: A Novel Deep Neural Network Architecture for Multi-Task Learning","date":"2017-10-27","arxiv_id":"1710.10036","repositories_listed":1,"syntology":null},{"url":"/paper/learning-approximate-stochastic-transition","slug":"learning-approximate-stochastic-transition","title":"Learning Approximate Stochastic Transition Models","date":"2017-10-26","arxiv_id":"1710.09718","repositories_listed":1,"syntology":null},{"url":"/paper/decomposition-of-uncertainty-in-bayesian-deep","slug":"decomposition-of-uncertainty-in-bayesian-deep","title":"Decomposition of Uncertainty in Bayesian Deep Learning for Efficient and Risk-sensitive Learning","date":"2017-10-19","arxiv_id":"1710.07283","repositories_listed":1,"syntology":null},{"url":"/paper/the-effects-of-memory-replay-in-reinforcement","slug":"the-effects-of-memory-replay-in-reinforcement","title":"The Effects of Memory Replay in Reinforcement Learning","date":"2017-10-18","arxiv_id":"1710.06574","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-deep-execution-monitoring","slug":"vision-based-deep-execution-monitoring","title":"Vision-based deep execution monitoring","date":"2017-09-29","arxiv_id":"1709.10507","repositories_listed":1,"syntology":null},{"url":"/paper/learning-complex-dexterous-manipulation-with","slug":"learning-complex-dexterous-manipulation-with","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","date":"2017-09-28","arxiv_id":"1709.10087","repositories_listed":1,"syntology":null},{"url":"/paper/cold-start-reinforcement-learning-with","slug":"cold-start-reinforcement-learning-with","title":"Cold-Start Reinforcement Learning with Softmax Policy Gradient","date":"2017-09-27","arxiv_id":"1709.09346","repositories_listed":1,"syntology":null},{"url":"/paper/exposure-a-white-box-photo-post-processing","slug":"exposure-a-white-box-photo-post-processing","title":"Exposure: A White-Box Photo Post-Processing Framework","date":"2017-09-27","arxiv_id":"1709.09602","repositories_listed":1,"syntology":null},{"url":"/paper/mdp-environments-for-the-openai-gym","slug":"mdp-environments-for-the-openai-gym","title":"MDP environments for the OpenAI Gym","date":"2017-09-26","arxiv_id":"1709.09069","repositories_listed":1,"syntology":null},{"url":"/paper/optiongan-learning-joint-reward-policy","slug":"optiongan-learning-joint-reward-policy","title":"OptionGAN: Learning Joint Reward-Policy Options using Generative Adversarial Inverse Reinforcement Learning","date":"2017-09-20","arxiv_id":"1709.06683","repositories_listed":1,"syntology":null},{"url":"/paper/using-parameterized-black-box-priors-to-scale","slug":"using-parameterized-black-box-priors-to-scale","title":"Using Parameterized Black-Box Priors to Scale Up Model-Based Policy Search for Robotics","date":"2017-09-20","arxiv_id":"1709.06917","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-event-driven","slug":"deep-reinforcement-learning-for-event-driven","title":"Deep Reinforcement Learning for Event-Driven Multi-Agent Decision Processes","date":"2017-09-19","arxiv_id":"1709.06656","repositories_listed":1,"syntology":null},{"url":"/paper/guided-deep-reinforcement-learning-for-swarm","slug":"guided-deep-reinforcement-learning-for-swarm","title":"Guided Deep Reinforcement Learning for Swarm Systems","date":"2017-09-18","arxiv_id":"1709.06011","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-deep-reinforcement-learning-for-swarm#ran","syntology_url":"https://syntology.ai/paper/1709.06011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.06011"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for","slug":"deep-reinforcement-learning-for","title":"Deep Reinforcement Learning for Conversational AI","date":"2017-09-15","arxiv_id":"1709.05067","repositories_listed":1,"syntology":null},{"url":"/paper/shapechanger-environments-for-transfer","slug":"shapechanger-environments-for-transfer","title":"Shapechanger: Environments for Transfer Learning","date":"2017-09-15","arxiv_id":"1709.05070","repositories_listed":1,"syntology":null},{"url":"/paper/the-uncertainty-bellman-equation-and","slug":"the-uncertainty-bellman-equation-and","title":"The Uncertainty Bellman Equation and Exploration","date":"2017-09-15","arxiv_id":"1709.05380","repositories_listed":1,"syntology":null},{"url":"/paper/automated-cloud-provisioning-on-aws-using","slug":"automated-cloud-provisioning-on-aws-using","title":"Automated Cloud Provisioning on AWS using Deep Reinforcement Learning","date":"2017-09-13","arxiv_id":"1709.04305","repositories_listed":1,"syntology":null},{"url":"/paper/stack-captioning-coarse-to-fine-learning-for","slug":"stack-captioning-coarse-to-fine-learning-for","title":"Stack-Captioning: Coarse-to-Fine Learning for Image Captioning","date":"2017-09-11","arxiv_id":"1709.03376","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-bandits-balancing-the-exploration","slug":"bayesian-bandits-balancing-the-exploration","title":"Bayesian bandits: balancing the exploration-exploitation tradeoff via double sampling","date":"2017-09-10","arxiv_id":"1709.03162","repositories_listed":1,"syntology":null},{"url":"/paper/variational-inference-for-the-multi-armed","slug":"variational-inference-for-the-multi-armed","title":"Variational inference for the multi-armed contextual bandit","date":"2017-09-10","arxiv_id":"1709.03163","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-descent-search-and-its-acceleration","slug":"mirror-descent-search-and-its-acceleration","title":"Mirror Descent Search and its Acceleration","date":"2017-09-08","arxiv_id":"1709.02535","repositories_listed":1,"syntology":null},{"url":"/paper/prosocial-learning-agents-solve-generalized","slug":"prosocial-learning-agents-solve-generalized","title":"Prosocial learning agents solve generalized Stag Hunts better than selfish ones","date":"2017-09-08","arxiv_id":"1709.02865","repositories_listed":1,"syntology":null},{"url":"/paper/speeding-up-reinforcement-learning-based","slug":"speeding-up-reinforcement-learning-based","title":"Speeding up Reinforcement Learning-based Information Extraction Training using Asynchronous Methods","date":"2017-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/r3-reinforced-reader-ranker-for-open-domain","slug":"r3-reinforced-reader-ranker-for-open-domain","title":"R$^3$: Reinforced Reader-Ranker for Open-Domain Question Answering","date":"2017-08-31","arxiv_id":"1709.00023","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-via-shielding","slug":"safe-reinforcement-learning-via-shielding","title":"Safe Reinforcement Learning via Shielding","date":"2017-08-29","arxiv_id":"1708.08611","repositories_listed":1,"syntology":null},{"url":"/paper/deep-object-centric-representations-for","slug":"deep-object-centric-representations-for","title":"Deep Object-Centric Representations for Generalizable Robot Learning","date":"2017-08-14","arxiv_id":"1708.04225","repositories_listed":1,"syntology":null},{"url":"/paper/group-driven-reinforcement-learning-for","slug":"group-driven-reinforcement-learning-for","title":"Group-driven Reinforcement Learning for Personalized mHealth Intervention","date":"2017-08-14","arxiv_id":"1708.04001","repositories_listed":1,"syntology":null},{"url":"/paper/reproducibility-of-benchmarked-deep","slug":"reproducibility-of-benchmarked-deep","title":"Reproducibility of Benchmarked Deep Reinforcement Learning Tasks for Continuous Control","date":"2017-08-10","arxiv_id":"1708.04133","repositories_listed":1,"syntology":null},{"url":"/paper/learning-how-to-active-learn-a-deep","slug":"learning-how-to-active-learn-a-deep","title":"Learning how to Active Learn: A Deep Reinforcement Learning Approach","date":"2017-08-08","arxiv_id":"1708.02383","repositories_listed":1,"syntology":null},{"url":"/paper/stardata-a-starcraft-ai-research-dataset","slug":"stardata-a-starcraft-ai-research-dataset","title":"STARDATA: A StarCraft AI Research Dataset","date":"2017-08-07","arxiv_id":"1708.02139","repositories_listed":1,"syntology":null},{"url":"/paper/variational-generative-stochastic-networks","slug":"variational-generative-stochastic-networks","title":"Variational Generative Stochastic Networks with Collaborative Shaping","date":"2017-08-02","arxiv_id":"1708.00805","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-language-for-transfer-in-deep","slug":"grounding-language-for-transfer-in-deep","title":"Grounding Language for Transfer in Deep Reinforcement Learning","date":"2017-08-01","arxiv_id":"1708.00133","repositories_listed":1,"syntology":null},{"url":"/paper/darla-improving-zero-shot-transfer-in","slug":"darla-improving-zero-shot-transfer-in","title":"DARLA: Improving Zero-Shot Transfer in Reinforcement Learning","date":"2017-07-26","arxiv_id":"1707.08475","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-multi-task-learning","slug":"a-survey-on-multi-task-learning","title":"A Survey on Multi-Task Learning","date":"2017-07-25","arxiv_id":"1707.08114","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-bandit-neural","slug":"reinforcement-learning-for-bandit-neural","title":"Reinforcement Learning for Bandit Neural Machine Translation with Simulated Human Feedback","date":"2017-07-24","arxiv_id":"1707.07402","repositories_listed":1,"syntology":null},{"url":"/paper/trial-without-error-towards-safe","slug":"trial-without-error-towards-safe","title":"Trial without Error: Towards Safe Reinforcement Learning via Human Intervention","date":"2017-07-17","arxiv_id":"1707.05173","repositories_listed":1,"syntology":null},{"url":"/paper/lenient-multi-agent-deep-reinforcement","slug":"lenient-multi-agent-deep-reinforcement","title":"Lenient Multi-Agent Deep Reinforcement Learning","date":"2017-07-14","arxiv_id":"1707.04402","repositories_listed":1,"syntology":null},{"url":"/paper/merge-or-not-learning-to-group-faces-via","slug":"merge-or-not-learning-to-group-faces-via","title":"Merge or Not? Learning to Group Faces via Imitation Learning","date":"2017-07-13","arxiv_id":"1707.03986","repositories_listed":1,"syntology":null},{"url":"/paper/representation-learning-for-grounded-spatial","slug":"representation-learning-for-grounded-spatial","title":"Representation Learning for Grounded Spatial Reasoning","date":"2017-07-13","arxiv_id":"1707.03938","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-from-observation-learning-to","slug":"imitation-from-observation-learning-to","title":"Imitation from Observation: Learning to Imitate Behaviors from Raw Video via Context Translation","date":"2017-07-11","arxiv_id":"1707.03374","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-behaviors-from-motion-capture","slug":"learning-human-behaviors-from-motion-capture","title":"Learning human behaviors from motion capture by adversarial imitation","date":"2017-07-07","arxiv_id":"1707.02201","repositories_listed":1,"syntology":null},{"url":"/paper/trust-pcl-an-off-policy-trust-region-method","slug":"trust-pcl-an-off-policy-trust-region-method","title":"Trust-PCL: An Off-Policy Trust Region Method for Continuous Control","date":"2017-07-06","arxiv_id":"1707.01891","repositories_listed":1,"syntology":null},{"url":"/paper/maintaining-cooperation-in-complex-social","slug":"maintaining-cooperation-in-complex-social","title":"Maintaining cooperation in complex social dilemmas using deep reinforcement learning","date":"2017-07-04","arxiv_id":"1707.01068","repositories_listed":1,"syntology":null},{"url":"/paper/action-decision-networks-for-visual-tracking","slug":"action-decision-networks-for-visual-tracking","title":"Action-Decision Networks for Visual Tracking With Deep Reinforcement Learning","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-sequence-model-training-via-divergence","slug":"neural-sequence-model-training-via-divergence","title":"Neural Sequence Model Training via $α$-divergence Minimization","date":"2017-06-30","arxiv_id":"1706.10031","repositories_listed":1,"syntology":null},{"url":"/paper/neural-slam-learning-to-explore-with-external","slug":"neural-slam-learning-to-explore-with-external","title":"Neural SLAM: Learning to Explore with External Memory","date":"2017-06-29","arxiv_id":"1706.09520","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-in-feature-space-for","slug":"count-based-exploration-in-feature-space-for","title":"Count-Based Exploration in Feature Space for Reinforcement Learning","date":"2017-06-25","arxiv_id":"1706.08090","repositories_listed":1,"syntology":null},{"url":"/paper/a-self-adaptive-proposal-model-for-temporal","slug":"a-self-adaptive-proposal-model-for-temporal","title":"A Self-Adaptive Proposal Model for Temporal Action Detection based on Reinforcement Learning","date":"2017-06-22","arxiv_id":"1706.07251","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-reinforcement-learning-with","slug":"data-efficient-reinforcement-learning-with","title":"Data-Efficient Reinforcement Learning with Probabilistic Model Predictive Control","date":"2017-06-20","arxiv_id":"1706.06491","repositories_listed":1,"syntology":null},{"url":"/paper/dex-incremental-learning-for-complex","slug":"dex-incremental-learning-for-complex","title":"Dex: Incremental Learning for Complex Environments in Deep Reinforcement Learning","date":"2017-06-19","arxiv_id":"1706.05749","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-task-generalization-with-multi-task","slug":"zero-shot-task-generalization-with-multi-task","title":"Zero-Shot Task Generalization with Multi-Task Deep Reinforcement Learning","date":"2017-06-15","arxiv_id":"1706.05064","repositories_listed":1,"syntology":null},{"url":"/paper/device-placement-optimization-with","slug":"device-placement-optimization-with","title":"Device Placement Optimization with Reinforcement Learning","date":"2017-06-13","arxiv_id":"1706.04972","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-reward-architecture-for-reinforcement","slug":"hybrid-reward-architecture-for-reinforcement","title":"Hybrid Reward Architecture for Reinforcement Learning","date":"2017-06-13","arxiv_id":"1706.04208","repositories_listed":1,"syntology":null},{"url":"/paper/implications-of-decentralized-q-learning","slug":"implications-of-decentralized-q-learning","title":"Implications of Decentralized Q-learning Resource Allocation in Wireless Networks","date":"2017-05-30","arxiv_id":"1705.10508","repositories_listed":1,"syntology":null},{"url":"/paper/objective-reinforced-generative-adversarial","slug":"objective-reinforced-generative-adversarial","title":"Objective-Reinforced Generative Adversarial Networks (ORGAN) for Sequence Generation Models","date":"2017-05-30","arxiv_id":"1705.10843","repositories_listed":1,"syntology":null},{"url":"/paper/universal-reinforcement-learning-algorithms","slug":"universal-reinforcement-learning-algorithms","title":"Universal Reinforcement Learning Algorithms: Survey and Experiments","date":"2017-05-30","arxiv_id":"1705.10557","repositories_listed":1,"syntology":null},{"url":"/paper/free-energy-based-reinforcement-learning","slug":"free-energy-based-reinforcement-learning","title":"Free energy-based reinforcement learning using a quantum processor","date":"2017-05-29","arxiv_id":"1706.00074","repositories_listed":1,"syntology":null},{"url":"/paper/latent-intention-dialogue-models","slug":"latent-intention-dialogue-models","title":"Latent Intention Dialogue Models","date":"2017-05-29","arxiv_id":"1705.10229","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-corrupted","slug":"reinforcement-learning-with-a-corrupted","title":"Reinforcement Learning with a Corrupted Reward Channel","date":"2017-05-23","arxiv_id":"1705.08417","repositories_listed":1,"syntology":null},{"url":"/paper/safe-model-based-reinforcement-learning-with","slug":"safe-model-based-reinforcement-learning-with","title":"Safe Model-based Reinforcement Learning with Stability Guarantees","date":"2017-05-23","arxiv_id":"1705.08551","repositories_listed":1,"syntology":null},{"url":"/paper/aixijs-a-software-demo-for-general","slug":"aixijs-a-software-demo-for-general","title":"AIXIjs: A Software Demo for General Reinforcement Learning","date":"2017-05-22","arxiv_id":"1705.07615","repositories_listed":1,"syntology":null},{"url":"/paper/guide-actor-critic-for-continuous-control","slug":"guide-actor-critic-for-continuous-control","title":"Guide Actor-Critic for Continuous Control","date":"2017-05-22","arxiv_id":"1705.07606","repositories_listed":1,"syntology":null},{"url":"/paper/feature-control-as-intrinsic-motivation-for","slug":"feature-control-as-intrinsic-motivation-for","title":"Feature Control as Intrinsic Motivation for Hierarchical Reinforcement Learning","date":"2017-05-18","arxiv_id":"1705.06769","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-goal-generation-for-reinforcement","slug":"automatic-goal-generation-for-reinforcement","title":"Automatic Goal Generation for Reinforcement Learning Agents","date":"2017-05-17","arxiv_id":"1705.06366","repositories_listed":1,"syntology":null},{"url":"/paper/integral-policy-iterations-for-reinforcement","slug":"integral-policy-iterations-for-reinforcement","title":"Policy Iterations for Reinforcement Learning Problems in Continuous Time and Space -- Fundamental Theory and Methods","date":"2017-05-09","arxiv_id":"1705.03520","repositories_listed":1,"syntology":null},{"url":"/paper/metacontrol-for-adaptive-imagination-based","slug":"metacontrol-for-adaptive-imagination-based","title":"Metacontrol for Adaptive Imagination-Based Optimization","date":"2017-05-07","arxiv_id":"1705.02670","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multimodal-transition-dynamics-for","slug":"learning-multimodal-transition-dynamics-for","title":"Learning Multimodal Transition Dynamics for Model-Based Reinforcement Learning","date":"2017-05-01","arxiv_id":"1705.00470","repositories_listed":1,"syntology":null},{"url":"/paper/mapping-instructions-and-visual-observations","slug":"mapping-instructions-and-visual-observations","title":"Mapping Instructions and Visual Observations to Actions with Reinforcement Learning","date":"2017-04-28","arxiv_id":"1704.08795","repositories_listed":1,"syntology":null},{"url":"/paper/on-improving-deep-reinforcement-learning-for","slug":"on-improving-deep-reinforcement-learning-for","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2017-04-26","arxiv_id":"1704.07978","repositories_listed":1,"syntology":null},{"url":"/paper/modular-multi-objective-deep-reinforcement","slug":"modular-multi-objective-deep-reinforcement","title":"Modular Multi-Objective Deep Reinforcement Learning with Decision Values","date":"2017-04-21","arxiv_id":"1704.06676","repositories_listed":1,"syntology":null},{"url":"/paper/beating-atari-with-natural-language-guided","slug":"beating-atari-with-natural-language-guided","title":"Beating Atari with Natural Language Guided Reinforcement Learning","date":"2017-04-18","arxiv_id":"1704.05539","repositories_listed":1,"syntology":null},{"url":"/paper/muse-modularizing-unsupervised-sense","slug":"muse-modularizing-unsupervised-sense","title":"MUSE: Modularizing Unsupervised Sense Embeddings","date":"2017-04-15","arxiv_id":"1704.04601","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-differentiable-relaxations-of","slug":"optimizing-differentiable-relaxations-of","title":"Optimizing Differentiable Relaxations of Coreference Evaluation Metrics","date":"2017-04-14","arxiv_id":"1704.04451","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-framework-for","slug":"deep-reinforcement-learning-framework-for","title":"Deep Reinforcement Learning framework for Autonomous Driving","date":"2017-04-08","arxiv_id":"1704.02532","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-simplification-with-deep","slug":"sentence-simplification-with-deep","title":"Sentence Simplification with Deep Reinforcement Learning","date":"2017-03-31","arxiv_id":"1703.10931","repositories_listed":1,"syntology":null},{"url":"/paper/faster-reinforcement-learning-using-active","slug":"faster-reinforcement-learning-using-active","title":"Faster Reinforcement Learning Using Active Simulators","date":"2017-03-22","arxiv_id":"1703.07853","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-pac-and-regret-uniform-pac-bounds","slug":"unifying-pac-and-regret-uniform-pac-bounds","title":"Unifying PAC and Regret: Uniform PAC Bounds for Episodic Reinforcement Learning","date":"2017-03-22","arxiv_id":"1703.07710","repositories_listed":1,"syntology":null},{"url":"/paper/black-box-data-efficient-policy-search-for","slug":"black-box-data-efficient-policy-search-for","title":"Black-Box Data-efficient Policy Search for Robotics","date":"2017-03-21","arxiv_id":"1703.07261","repositories_listed":1,"syntology":null},{"url":"/paper/towards-diverse-and-natural-image","slug":"towards-diverse-and-natural-image","title":"Towards Diverse and Natural Image Descriptions via a Conditional GAN","date":"2017-03-17","arxiv_id":"1703.06029","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-regret-bounds-for-reinforcement","slug":"minimax-regret-bounds-for-reinforcement","title":"Minimax Regret Bounds for Reinforcement Learning","date":"2017-03-16","arxiv_id":"1703.05449","repositories_listed":1,"syntology":null},{"url":"/paper/deep-variation-structured-reinforcement","slug":"deep-variation-structured-reinforcement","title":"Deep Variation-structured Reinforcement Learning for Visual Relationship and Attribute Detection","date":"2017-03-08","arxiv_id":"1703.03054","repositories_listed":1,"syntology":null}],"record_sha256":"37969bed36c1f464b17ab99e5417362c559175d33abf009fb4103b76ad5ed92c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}