{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/40","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":40,"pages_in_order":132,"rows_per_page":100,"rows":[3901,4000],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/39","next":"/task/reinforcement-learning/papers/41","papers":[{"url":"/paper/market-self-learning-of-signals-impact-and","slug":"market-self-learning-of-signals-impact-and","title":"Market Self-Learning of Signals, Impact and Optimal Trading: Invisible Hand Inference with Free Energy","date":"2018-05-16","arxiv_id":"1805.06126","repositories_listed":1,"syntology":null},{"url":"/paper/progress-compress-a-scalable-framework-for","slug":"progress-compress-a-scalable-framework-for","title":"Progress & Compress: A scalable framework for continual learning","date":"2018-05-16","arxiv_id":"1805.06370","repositories_listed":1,"syntology":null},{"url":"/paper/do-deep-reinforcement-learning-agents-model","slug":"do-deep-reinforcement-learning-agents-model","title":"Do deep reinforcement learning agents model intentions?","date":"2018-05-15","arxiv_id":"1805.06020","repositories_listed":1,"syntology":null},{"url":"/paper/unpaired-sentiment-to-sentiment-translation-a","slug":"unpaired-sentiment-to-sentiment-translation-a","title":"Unpaired Sentiment-to-Sentiment Translation: A Cycled Reinforcement Learning Approach","date":"2018-05-14","arxiv_id":"1805.05181","repositories_listed":1,"syntology":null},{"url":"/paper/gan-q-learning","slug":"gan-q-learning","title":"GAN Q-learning","date":"2018-05-13","arxiv_id":"1805.04874","repositories_listed":1,"syntology":null},{"url":"/paper/general-solutions-for-nonlinear-differential","slug":"general-solutions-for-nonlinear-differential","title":"General solutions for nonlinear differential equations: a rule-based self-learning approach using deep reinforcement learning","date":"2018-05-13","arxiv_id":"1805.07297","repositories_listed":1,"syntology":null},{"url":"/paper/backpropagating-through-structured-argmax","slug":"backpropagating-through-structured-argmax","title":"Backpropagating through Structured Argmax using a SPIGOT","date":"2018-05-12","arxiv_id":"1805.04658","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-reinforcement-learning-for","slug":"end-to-end-reinforcement-learning-for","title":"End-to-End Reinforcement Learning for Automatic Taxonomy Induction","date":"2018-05-10","arxiv_id":"1805.04044","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1805.04044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.04044"}},"official":{"repos":["morningmoni/TaxoRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-estimation-for-variance-reduction-in","slug":"reward-estimation-for-variance-reduction-in","title":"Reward Estimation for Variance Reduction in Deep Reinforcement Learning","date":"2018-05-09","arxiv_id":"1805.03359","repositories_listed":1,"syntology":null},{"url":"/paper/ffnet-video-fast-forwarding-via-reinforcement","slug":"ffnet-video-fast-forwarding-via-reinforcement","title":"FFNet: Video Fast-Forwarding via Reinforcement Learning","date":"2018-05-08","arxiv_id":"1805.02792","repositories_listed":1,"syntology":null},{"url":"/paper/polite-dialogue-generation-without-parallel","slug":"polite-dialogue-generation-without-parallel","title":"Polite Dialogue Generation Without Parallel Data","date":"2018-05-08","arxiv_id":"1805.03162","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polite-dialogue-generation-without-parallel#ran","syntology_url":"https://syntology.ai/paper/1805.03162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.03162"}},"official":null}},{"url":"/paper/ranking-for-relevance-and-display-preferences","slug":"ranking-for-relevance-and-display-preferences","title":"Ranking for Relevance and Display Preferences in Complex Presentation Layouts","date":"2018-05-07","arxiv_id":"1805.02404","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to","slug":"a-reinforcement-learning-approach-to","title":"A Reinforcement Learning Approach to Interactive-Predictive Neural Machine Translation","date":"2018-05-03","arxiv_id":"1805.01553","repositories_listed":1,"syntology":null},{"url":"/paper/vine-an-open-source-interactive-data","slug":"vine-an-open-source-interactive-data","title":"VINE: An Open Source Interactive Data Visualization Tool for Neuroevolution","date":"2018-05-03","arxiv_id":"1805.01141","repositories_listed":1,"syntology":null},{"url":"/paper/dialog-based-interactive-image-retrieval","slug":"dialog-based-interactive-image-retrieval","title":"Dialog-based Interactive Image Retrieval","date":"2018-05-01","arxiv_id":"1805.00145","repositories_listed":1,"syntology":null},{"url":"/paper/a-tree-search-algorithm-for-sequence-labeling","slug":"a-tree-search-algorithm-for-sequence-labeling","title":"A Tree Search Algorithm for Sequence Labeling","date":"2018-04-29","arxiv_id":"1804.10911","repositories_listed":1,"syntology":null},{"url":"/paper/from-credit-assignment-to-entropy","slug":"from-credit-assignment-to-entropy","title":"From Credit Assignment to Entropy Regularization: Two New Algorithms for Neural Sequence Prediction","date":"2018-04-29","arxiv_id":"1804.10974","repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-dynamics-and-reward-for-transfer","slug":"decoupling-dynamics-and-reward-for-transfer","title":"Decoupling Dynamics and Reward for Transfer Learning","date":"2018-04-27","arxiv_id":"1804.10689","repositories_listed":1,"syntology":null},{"url":"/paper/crawling-in-rogues-dungeons-with-partitioned","slug":"crawling-in-rogues-dungeons-with-partitioned","title":"Crawling in Rogue's dungeons with (partitioned) A3C","date":"2018-04-23","arxiv_id":"1804.08685","repositories_listed":1,"syntology":null},{"url":"/paper/towards-symbolic-reinforcement-learning-with","slug":"towards-symbolic-reinforcement-learning-with","title":"Towards Symbolic Reinforcement Learning with Common Sense","date":"2018-04-23","arxiv_id":"1804.08597","repositories_listed":1,"syntology":null},{"url":"/paper/a-stable-and-effective-learning-strategy-for","slug":"a-stable-and-effective-learning-strategy-for","title":"A Stable and Effective Learning Strategy for Trainable Greedy Decoding","date":"2018-04-21","arxiv_id":"1804.07915","repositories_listed":1,"syntology":null},{"url":"/paper/lipschitz-continuity-in-model-based","slug":"lipschitz-continuity-in-model-based","title":"Lipschitz Continuity in Model-based Reinforcement Learning","date":"2018-04-19","arxiv_id":"1804.07193","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-on-overfitting-in-deep-reinforcement","slug":"a-study-on-overfitting-in-deep-reinforcement","title":"A Study on Overfitting in Deep Reinforcement Learning","date":"2018-04-18","arxiv_id":"1804.06893","repositories_listed":1,"syntology":null},{"url":"/paper/dialogue-learning-with-human-teaching-and","slug":"dialogue-learning-with-human-teaching-and","title":"Dialogue Learning with Human Teaching and Feedback in End-to-End Trainable Task-Oriented Dialogue Systems","date":"2018-04-18","arxiv_id":"1804.06512","repositories_listed":1,"syntology":null},{"url":"/paper/cytonrl-an-efficient-reinforcement-learning","slug":"cytonrl-an-efficient-reinforcement-learning","title":"CytonRL: an Efficient Reinforcement Learning Open-source Toolkit Implemented in C++","date":"2018-04-14","arxiv_id":"1804.05834","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-linguistic-communication-from","slug":"emergence-of-linguistic-communication-from","title":"Emergence of Linguistic Communication from Referential Games with Symbolic and Pixel Input","date":"2018-04-11","arxiv_id":"1804.03984","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergence-of-linguistic-communication-from#ran","syntology_url":"https://syntology.ai/paper/1804.03984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.03984"}},"official":null}},{"url":"/paper/emergent-communication-through-negotiation","slug":"emergent-communication-through-negotiation","title":"Emergent Communication through Negotiation","date":"2018-04-11","arxiv_id":"1804.03980","repositories_listed":1,"syntology":null},{"url":"/paper/market-making-via-reinforcement-learning","slug":"market-making-via-reinforcement-learning","title":"Market Making via Reinforcement Learning","date":"2018-04-11","arxiv_id":"1804.04216","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-communications-systems","slug":"end-to-end-learning-of-communications-systems","title":"End-to-End Learning of Communications Systems Without a Channel Model","date":"2018-04-06","arxiv_id":"1804.02276","repositories_listed":1,"syntology":null},{"url":"/paper/starcraft-micromanagement-with-reinforcement","slug":"starcraft-micromanagement-with-reinforcement","title":"StarCraft Micromanagement with Reinforcement Learning and Curriculum Transfer Learning","date":"2018-04-03","arxiv_id":"1804.00810","repositories_listed":1,"syntology":null},{"url":"/paper/universal-planning-networks","slug":"universal-planning-networks","title":"Universal Planning Networks","date":"2018-04-02","arxiv_id":"1804.00645","repositories_listed":1,"syntology":null},{"url":"/paper/towards-learning-transferable-conversational","slug":"towards-learning-transferable-conversational","title":"Towards Learning Transferable Conversational Skills using Multi-dimensional Dialogue Modelling","date":"2018-03-31","arxiv_id":"1804.00146","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-predictive-memory-in-a-goal","slug":"unsupervised-predictive-memory-in-a-goal","title":"Unsupervised Predictive Memory in a Goal-Directed Agent","date":"2018-03-28","arxiv_id":"1803.10760","repositories_listed":1,"syntology":null},{"url":"/paper/neuronal-circuit-policies","slug":"neuronal-circuit-policies","title":"Neuronal Circuit Policies","date":"2018-03-22","arxiv_id":"1803.08554","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-video-captioning-with-multitask","slug":"end-to-end-video-captioning-with-multitask","title":"End-to-End Video Captioning with Multitask Reinforcement Learning","date":"2018-03-21","arxiv_id":"1803.07950","repositories_listed":1,"syntology":null},{"url":"/paper/look-before-you-leap-bridging-model-free-and","slug":"look-before-you-leap-bridging-model-free-and","title":"Look Before You Leap: Bridging Model-Free and Model-Based Reinforcement Learning for Planned-Ahead Vision-and-Language Navigation","date":"2018-03-21","arxiv_id":"1803.07729","repositories_listed":1,"syntology":null},{"url":"/paper/automated-curriculum-learning-by-rewarding","slug":"automated-curriculum-learning-by-rewarding","title":"Automated Curriculum Learning by Rewarding Temporally Rare Events","date":"2018-03-19","arxiv_id":"1803.07131","repositories_listed":1,"syntology":null},{"url":"/paper/composable-deep-reinforcement-learning-for","slug":"composable-deep-reinforcement-learning-for","title":"Composable Deep Reinforcement Learning for Robotic Manipulation","date":"2018-03-19","arxiv_id":"1803.06773","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-general-video-games-via-an","slug":"learning-to-play-general-video-games-via-an","title":"Learning to Play General Video-Games via an Object Embedding Network","date":"2018-03-14","arxiv_id":"1803.05262","repositories_listed":1,"syntology":null},{"url":"/paper/the-advantage-of-doubling-a-deep","slug":"the-advantage-of-doubling-a-deep","title":"The Advantage of Doubling: A Deep Reinforcement Learning Approach to Studying the Double Team in the NBA","date":"2018-03-08","arxiv_id":"1803.02940","repositories_listed":1,"syntology":null},{"url":"/paper/discontinuity-sensitive-optimal-control","slug":"discontinuity-sensitive-optimal-control","title":"Discontinuity-Sensitive Optimal Control Learning by Mixture of Experts","date":"2018-03-07","arxiv_id":"1803.02493","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-to-rank-in-e-commerce","slug":"reinforcement-learning-to-rank-in-e-commerce","title":"Reinforcement Learning to Rank in E-Commerce Search Engine: Formalization, Analysis, and Application","date":"2018-03-02","arxiv_id":"1803.00710","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-vision-based","slug":"deep-reinforcement-learning-for-vision-based","title":"Deep Reinforcement Learning for Vision-Based Robotic Grasping: A Simulated Comparative Evaluation of Off-Policy Methods","date":"2018-02-28","arxiv_id":"1802.10264","repositories_listed":1,"syntology":null},{"url":"/paper/selective-experience-replay-for-lifelong","slug":"selective-experience-replay-for-lifelong","title":"Selective Experience Replay for Lifelong Learning","date":"2018-02-28","arxiv_id":"1802.10269","repositories_listed":1,"syntology":null},{"url":"/paper/identification-of-ltv-dynamical-models-with","slug":"identification-of-ltv-dynamical-models-with","title":"Identification of LTV Dynamical Models with Smooth or Discontinuous Time Evolution by means of Convex Optimization","date":"2018-02-27","arxiv_id":"1802.09794","repositories_listed":1,"syntology":null},{"url":"/paper/the-mirage-of-action-dependent-baselines-in","slug":"the-mirage-of-action-dependent-baselines-in","title":"The Mirage of Action-Dependent Baselines in Reinforcement Learning","date":"2018-02-27","arxiv_id":"1802.10031","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-mirage-of-action-dependent-baselines-in#ran","syntology_url":"https://syntology.ai/paper/1802.10031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10031"}},"official":null}},{"url":"/paper/modeling-others-using-oneself-in-multi-agent","slug":"modeling-others-using-oneself-in-multi-agent","title":"Modeling Others using Oneself in Multi-Agent Reinforcement Learning","date":"2018-02-26","arxiv_id":"1802.09640","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-and-imitation-learning-for","slug":"reinforcement-and-imitation-learning-for","title":"Reinforcement and Imitation Learning for Diverse Visuomotor Skills","date":"2018-02-26","arxiv_id":"1802.09564","repositories_listed":1,"syntology":null},{"url":"/paper/back-to-basics-benchmarking-canonical","slug":"back-to-basics-benchmarking-canonical","title":"Back to Basics: Benchmarking Canonical Evolution Strategies for Playing Atari","date":"2018-02-24","arxiv_id":"1802.08842","repositories_listed":1,"syntology":null},{"url":"/paper/ranking-sentences-for-extractive","slug":"ranking-sentences-for-extractive","title":"Ranking Sentences for Extractive Summarization with Reinforcement Learning","date":"2018-02-23","arxiv_id":"1802.08636","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ranking-sentences-for-extractive#ran","syntology_url":"https://syntology.ai/paper/1802.08636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08636"}},"official":{"repos":["shashiongithub/Refresh"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/verifying-controllers-against-adversarial","slug":"verifying-controllers-against-adversarial","title":"Verifying Controllers Against Adversarial Examples with Bayesian Optimization","date":"2018-02-23","arxiv_id":"1802.08678","repositories_listed":1,"syntology":null},{"url":"/paper/structured-control-nets-for-deep","slug":"structured-control-nets-for-deep","title":"Structured Control Nets for Deep Reinforcement Learning","date":"2018-02-22","arxiv_id":"1802.08311","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-gather-without-communication","slug":"learning-to-gather-without-communication","title":"Learning to Gather without Communication","date":"2018-02-21","arxiv_id":"1802.07834","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-large-scale-fleet-management-via","slug":"efficient-large-scale-fleet-management-via","title":"Efficient Collaborative Multi-Agent Deep Reinforcement Learning for Large-Scale Fleet Management","date":"2018-02-18","arxiv_id":"1802.06444","repositories_listed":1,"syntology":null},{"url":"/paper/from-gameplay-to-symbolic-reasoning-learning","slug":"from-gameplay-to-symbolic-reasoning-learning","title":"From Gameplay to Symbolic Reasoning: Learning SAT Solver Heuristics in the Style of Alpha(Go) Zero","date":"2018-02-14","arxiv_id":"1802.05340","repositories_listed":1,"syntology":null},{"url":"/paper/gep-pg-decoupling-exploration-and","slug":"gep-pg-decoupling-exploration-and","title":"GEP-PG: Decoupling Exploration and Exploitation in Deep Reinforcement Learning Algorithms","date":"2018-02-14","arxiv_id":"1802.05054","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gep-pg-decoupling-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1802.05054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.05054"}},"official":{"repos":["flowersteam/geppg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-exploration-through-bayesian-deep-q","slug":"efficient-exploration-through-bayesian-deep-q","title":"Efficient Exploration through Bayesian Deep Q-Networks","date":"2018-02-13","arxiv_id":"1802.04412","repositories_listed":1,"syntology":null},{"url":"/paper/answerer-in-questioners-mind-information","slug":"answerer-in-questioners-mind-information","title":"Answerer in Questioner's Mind: Information Theoretic Approach to Goal-Oriented Visual Dialog","date":"2018-02-12","arxiv_id":"1802.03881","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-bias-span-constrained-exploration","slug":"efficient-bias-span-constrained-exploration","title":"Efficient Bias-Span-Constrained Exploration-Exploitation in Reinforcement Learning","date":"2018-02-12","arxiv_id":"1802.04020","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-model-based-deep-reinforcement","slug":"efficient-model-based-deep-reinforcement","title":"Efficient Model-Based Deep Reinforcement Learning with Variational State Tabulation","date":"2018-02-12","arxiv_id":"1802.04325","repositories_listed":1,"syntology":null},{"url":"/paper/state-representation-learning-for-control-an","slug":"state-representation-learning-for-control-an","title":"State Representation Learning for Control: An Overview","date":"2018-02-12","arxiv_id":"1802.04181","repositories_listed":1,"syntology":null},{"url":"/paper/more-robust-doubly-robust-off-policy","slug":"more-robust-doubly-robust-off-policy","title":"More Robust Doubly Robust Off-policy Evaluation","date":"2018-02-10","arxiv_id":"1802.03493","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-investigation-of-deep","slug":"a-critical-investigation-of-deep","title":"A Critical Investigation of Deep Reinforcement Learning for Navigation","date":"2018-02-07","arxiv_id":"1802.02274","repositories_listed":1,"syntology":null},{"url":"/paper/shared-autonomy-via-deep-reinforcement","slug":"shared-autonomy-via-deep-reinforcement","title":"Shared Autonomy via Deep Reinforcement Learning","date":"2018-02-06","arxiv_id":"1802.01744","repositories_listed":1,"syntology":null},{"url":"/paper/utility-decomposition-with-deep-corrections","slug":"utility-decomposition-with-deep-corrections","title":"Decomposition Methods with Deep Corrections for Reinforcement Learning","date":"2018-02-06","arxiv_id":"1802.01772","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-programming","slug":"deep-reinforcement-learning-for-programming","title":"Deep Reinforcement Learning for Programming Language Correction","date":"2018-01-31","arxiv_id":"1801.10467","repositories_listed":1,"syntology":null},{"url":"/paper/learning-the-reward-function-for-a","slug":"learning-the-reward-function-for-a","title":"Learning the Reward Function for a Misspecified Model","date":"2018-01-29","arxiv_id":"1801.09624","repositories_listed":1,"syntology":null},{"url":"/paper/active-neural-localization","slug":"active-neural-localization","title":"Active Neural Localization","date":"2018-01-24","arxiv_id":"1801.08214","repositories_listed":1,"syntology":null},{"url":"/paper/logically-constrained-reinforcement-learning","slug":"logically-constrained-reinforcement-learning","title":"Logically-Constrained Reinforcement Learning","date":"2018-01-24","arxiv_id":"1801.08099","repositories_listed":1,"syntology":null},{"url":"/paper/psychlab-a-psychology-laboratory-for-deep","slug":"psychlab-a-psychology-laboratory-for-deep","title":"Psychlab: A Psychology Laboratory for Deep Reinforcement Learning Agents","date":"2018-01-24","arxiv_id":"1801.08116","repositories_listed":1,"syntology":null},{"url":"/paper/learning-model-based-strategies-in-simple","slug":"learning-model-based-strategies-in-simple","title":"Learning model-based strategies in simple environments with hierarchical q-networks","date":"2018-01-20","arxiv_id":"1801.06689","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-deep-reinforcement-learning-learn","slug":"distributed-deep-reinforcement-learning-learn","title":"Distributed Deep Reinforcement Learning: Learn how to play Atari games in 21 minutes","date":"2018-01-09","arxiv_id":"1801.02852","repositories_listed":1,"syntology":null},{"url":"/paper/competitive-multi-agent-inverse-reinforcement","slug":"competitive-multi-agent-inverse-reinforcement","title":"Competitive Multi-agent Inverse Reinforcement Learning with Sub-optimal Demonstrations","date":"2018-01-07","arxiv_id":"1801.02124","repositories_listed":1,"syntology":null},{"url":"/paper/nervenet-learning-structured-policy-with","slug":"nervenet-learning-structured-policy-with","title":"NerveNet: Learning Structured Policy with Graph Neural Networks","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/parametrized-deep-q-networks-learning-playing","slug":"parametrized-deep-q-networks-learning-playing","title":"PARAMETRIZED DEEP Q-NETWORKS LEARNING: PLAYING ONLINE BATTLE ARENA WITH DISCRETE-CONTINUOUS HYBRID ACTION SPACE","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/residual-loss-prediction-reinforcement","slug":"residual-loss-prediction-reinforcement","title":"Residual Loss Prediction: Reinforcement Learning With No Incremental Feedback","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-structural-weight-uncertainty-for","slug":"learning-structural-weight-uncertainty-for","title":"Learning Structural Weight Uncertainty for Sequential Decision-Making","date":"2017-12-30","arxiv_id":"1801.00085","repositories_listed":1,"syntology":null},{"url":"/paper/f-divergence-constrained-policy-improvement","slug":"f-divergence-constrained-policy-improvement","title":"f-Divergence constrained policy improvement","date":"2017-12-29","arxiv_id":"1801.00056","repositories_listed":1,"syntology":null},{"url":"/paper/federated-control-with-hierarchical-multi","slug":"federated-control-with-hierarchical-multi","title":"Federated Control with Hierarchical Multi-Agent Deep Reinforcement Learning","date":"2017-12-22","arxiv_id":"1712.08266","repositories_listed":1,"syntology":null},{"url":"/paper/learning-intelligent-dialogs-for-bounding-box","slug":"learning-intelligent-dialogs-for-bounding-box","title":"Learning Intelligent Dialogs for Bounding Box Annotation","date":"2017-12-21","arxiv_id":"1712.08087","repositories_listed":1,"syntology":null},{"url":"/paper/safe-mutations-for-deep-and-recurrent-neural","slug":"safe-mutations-for-deep-and-recurrent-neural","title":"Safe Mutations for Deep and Recurrent Neural Networks through Output Gradients","date":"2017-12-18","arxiv_id":"1712.06563","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-text-generation-and-planning-for","slug":"hierarchical-text-generation-and-planning-for","title":"Hierarchical Text Generation and Planning for Strategic Dialogue","date":"2017-12-15","arxiv_id":"1712.05846","repositories_listed":1,"syntology":null},{"url":"/paper/dancin-seq2seq-fooling-text-classifiers-with","slug":"dancin-seq2seq-fooling-text-classifiers-with","title":"DANCin SEQ2SEQ: Fooling Text Classifiers with Adversarial Text Example Generation","date":"2017-12-14","arxiv_id":"1712.05419","repositories_listed":1,"syntology":null},{"url":"/paper/adaptation-to-criticality-through","slug":"adaptation-to-criticality-through","title":"Adaptation to criticality through organizational invariance in embodied agents","date":"2017-12-13","arxiv_id":"1712.05284","repositories_listed":1,"syntology":null},{"url":"/paper/qlbs-q-learner-in-the-black-scholes-merton","slug":"qlbs-q-learner-in-the-black-scholes-merton","title":"QLBS: Q-Learner in the Black-Scholes(-Merton) Worlds","date":"2017-12-13","arxiv_id":"1712.04609","repositories_listed":1,"syntology":null},{"url":"/paper/a-low-cost-ethics-shaping-approach-for","slug":"a-low-cost-ethics-shaping-approach-for","title":"A Low-Cost Ethics Shaping Approach for Designing Reinforcement Learning Agents","date":"2017-12-12","arxiv_id":"1712.04172","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-low-cost-ethics-shaping-approach-for#ran","syntology_url":"https://syntology.ai/paper/1712.04172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.04172"}},"official":null}},{"url":"/paper/assumed-density-filtering-q-learning","slug":"assumed-density-filtering-q-learning","title":"Assumed Density Filtering Q-learning","date":"2017-12-09","arxiv_id":"1712.03333","repositories_listed":1,"syntology":null},{"url":"/paper/iqa-visual-question-answering-in-interactive","slug":"iqa-visual-question-answering-in-interactive","title":"IQA: Visual Question Answering in Interactive Environments","date":"2017-12-09","arxiv_id":"1712.03316","repositories_listed":1,"syntology":null},{"url":"/paper/learning-a-generative-model-for-validity-in","slug":"learning-a-generative-model-for-validity-in","title":"Learning a Generative Model for Validity in Complex Discrete Structures","date":"2017-12-05","arxiv_id":"1712.01664","repositories_listed":1,"syntology":null},{"url":"/paper/runtime-neural-pruning","slug":"runtime-neural-pruning","title":"Runtime Neural Pruning","date":"2017-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/time-limits-in-reinforcement-learning","slug":"time-limits-in-reinforcement-learning","title":"Time Limits in Reinforcement Learning","date":"2017-12-01","arxiv_id":"1712.00378","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-de-novo-drug","slug":"deep-reinforcement-learning-for-de-novo-drug","title":"Deep Reinforcement Learning for De-Novo Drug Design","date":"2017-11-29","arxiv_id":"1711.10907","repositories_listed":1,"syntology":null},{"url":"/paper/crossmodal-attentive-skill-learner","slug":"crossmodal-attentive-skill-learner","title":"Crossmodal Attentive Skill Learner","date":"2017-11-28","arxiv_id":"1711.10314","repositories_listed":1,"syntology":null},{"url":"/paper/one-shot-reinforcement-learning-for-robot","slug":"one-shot-reinforcement-learning-for-robot","title":"One-Shot Reinforcement Learning for Robot Navigation with Interactive Replay","date":"2017-11-28","arxiv_id":"1711.10137","repositories_listed":1,"syntology":null},{"url":"/paper/plan-attend-generate-planning-for-sequence-to","slug":"plan-attend-generate-planning-for-sequence-to","title":"Plan, Attend, Generate: Planning for Sequence-to-Sequence Models","date":"2017-11-28","arxiv_id":"1711.10462","repositories_listed":1,"syntology":null},{"url":"/paper/risk-sensitive-inverse-reinforcement-learning","slug":"risk-sensitive-inverse-reinforcement-learning","title":"Risk-sensitive Inverse Reinforcement Learning via Semi- and Non-Parametric Methods","date":"2017-11-28","arxiv_id":"1711.10055","repositories_listed":1,"syntology":null},{"url":"/paper/divide-and-conquer-reinforcement-learning","slug":"divide-and-conquer-reinforcement-learning","title":"Divide-and-Conquer Reinforcement Learning","date":"2017-11-27","arxiv_id":"1711.09874","repositories_listed":1,"syntology":null},{"url":"/paper/generative-adversarial-network-for","slug":"generative-adversarial-network-for","title":"Generative Adversarial Network for Abstractive Text Summarization","date":"2017-11-26","arxiv_id":"1711.09357","repositories_listed":1,"syntology":null},{"url":"/paper/ethical-challenges-in-data-driven-dialogue","slug":"ethical-challenges-in-data-driven-dialogue","title":"Ethical Challenges in Data-Driven Dialogue Systems","date":"2017-11-24","arxiv_id":"1711.09050","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ethical-challenges-in-data-driven-dialogue#ran","syntology_url":"https://syntology.ai/paper/1711.09050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.09050"}},"official":{"repos":["Breakend/EthicsInDialogue"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/blockdrop-dynamic-inference-paths-in-residual","slug":"blockdrop-dynamic-inference-paths-in-residual","title":"BlockDrop: Dynamic Inference Paths in Residual Networks","date":"2017-11-22","arxiv_id":"1711.08393","repositories_listed":1,"syntology":null}],"record_sha256":"dc05550dd3d231812e8e59f8a5f3c71fca18461ea0c1721c093537f4e7edc1fd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}