{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/39","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":39,"pages_in_order":132,"rows_per_page":100,"rows":[3801,3900],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/38","next":"/task/reinforcement-learning/papers/40","papers":[{"url":"/paper/improving-reinforcement-learning-based-image","slug":"improving-reinforcement-learning-based-image","title":"Improving Reinforcement Learning Based Image Captioning with Natural Language Prior","date":"2018-09-13","arxiv_id":"1809.06227","repositories_listed":1,"syntology":null},{"url":"/paper/negative-update-intervals-in-deep-multi-agent","slug":"negative-update-intervals-in-deep-multi-agent","title":"Negative Update Intervals in Deep Multi-Agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05096","repositories_listed":1,"syntology":null},{"url":"/paper/combined-reinforcement-learning-via-abstract","slug":"combined-reinforcement-learning-via-abstract","title":"Combined Reinforcement Learning via Abstract Representations","date":"2018-09-12","arxiv_id":"1809.04506","repositories_listed":1,"syntology":null},{"url":"/paper/sai-a-sensible-artificial-intelligence-that","slug":"sai-a-sensible-artificial-intelligence-that","title":"SAI, a Sensible Artificial Intelligence that plays Go","date":"2018-09-11","arxiv_id":"1809.03928","repositories_listed":1,"syntology":null},{"url":"/paper/improving-optimization-bounds-using-machine","slug":"improving-optimization-bounds-using-machine","title":"Improving Optimization Bounds using Machine Learning: Decision Diagrams meet Deep Reinforcement Learning","date":"2018-09-10","arxiv_id":"1809.03359","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-optimization-bounds-using-machine#ran","syntology_url":"https://syntology.ai/paper/1809.03359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.03359"}},"official":{"repos":["qcappart/learning-DD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/keep-it-stupid-simple","slug":"keep-it-stupid-simple","title":"Combining imagination and heuristics to learn strategies that generalize","date":"2018-09-10","arxiv_id":"1809.03406","repositories_listed":1,"syntology":null},{"url":"/paper/learning-invariances-for-policy","slug":"learning-invariances-for-policy","title":"Learning Invariances for Policy Generalization","date":"2018-09-07","arxiv_id":"1809.02591","repositories_listed":1,"syntology":null},{"url":"/paper/archer-aggressive-rewards-to-counter-bias-in","slug":"archer-aggressive-rewards-to-counter-bias-in","title":"ARCHER: Aggressive Rewards to Counter bias in Hindsight Experience Replay","date":"2018-09-06","arxiv_id":"1809.02070","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-of-context-and-time-in","slug":"challenges-of-context-and-time-in","title":"Challenges of Context and Time in Reinforcement Learning: Introducing Space Fortress as a Benchmark","date":"2018-09-06","arxiv_id":"1809.02206","repositories_listed":1,"syntology":null},{"url":"/paper/accelerated-reinforcement-learning-for","slug":"accelerated-reinforcement-learning-for","title":"Accelerated Reinforcement Learning for Sentence Generation by Vocabulary Prediction","date":"2018-09-05","arxiv_id":"1809.01694","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-under-threats","slug":"reinforcement-learning-under-threats","title":"Reinforcement Learning under Threats","date":"2018-09-05","arxiv_id":"1809.01560","repositories_listed":1,"syntology":null},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exit-oos-towards-learning-from-planning-in","slug":"exit-oos-towards-learning-from-planning-in","title":"ExIt-OOS: Towards Learning from Planning in Imperfect Information Games","date":"2018-08-30","arxiv_id":"1808.10120","repositories_listed":1,"syntology":null},{"url":"/paper/april-interactively-learning-to-summarise-by","slug":"april-interactively-learning-to-summarise-by","title":"APRIL: Interactively Learning to Summarise by Combining Active Preference Learning and Reinforcement Learning","date":"2018-08-29","arxiv_id":"1808.09658","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/april-interactively-learning-to-summarise-by#ran","syntology_url":"https://syntology.ai/paper/1808.09658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09658"}},"official":{"repos":["UKPLab/emnlp2018-april"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cycle-of-learning-for-autonomous-systems-from","slug":"cycle-of-learning-for-autonomous-systems-from","title":"Cycle-of-Learning for Autonomous Systems from Human Interaction","date":"2018-08-28","arxiv_id":"1808.09572","repositories_listed":1,"syntology":null},{"url":"/paper/solar-deep-structured-representations-for","slug":"solar-deep-structured-representations-for","title":"SOLAR: Deep Structured Representations for Model-Based Reinforcement Learning","date":"2018-08-28","arxiv_id":"1808.09105","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-reinforcement-learning-for-neural","slug":"a-study-of-reinforcement-learning-for-neural","title":"A Study of Reinforcement Learning for Neural Machine Translation","date":"2018-08-27","arxiv_id":"1808.08866","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-study-of-reinforcement-learning-for-neural#ran","syntology_url":"https://syntology.ai/paper/1808.08866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.08866"}},"official":{"repos":["apeterswu/RL4NMT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-end-to-end-goal-oriented-dialog-with","slug":"learning-end-to-end-goal-oriented-dialog-with","title":"Learning End-to-End Goal-Oriented Dialog with Multiple Answers","date":"2018-08-24","arxiv_id":"1808.09996","repositories_listed":1,"syntology":null},{"url":"/paper/a-skeleton-based-model-for-promoting","slug":"a-skeleton-based-model-for-promoting","title":"A Skeleton-Based Model for Promoting Coherence Among Sentences in Narrative Story Generation","date":"2018-08-21","arxiv_id":"1808.06945","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-semantic-parsing-for-if-then","slug":"interactive-semantic-parsing-for-if-then","title":"Interactive Semantic Parsing for If-Then Recipes via Hierarchical Reinforcement Learning","date":"2018-08-21","arxiv_id":"1808.06740","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-visual-policy-network-for","slug":"context-aware-visual-policy-network-for","title":"Context-Aware Visual Policy Network for Sequence-Level Image Captioning","date":"2018-08-16","arxiv_id":"1808.05864","repositories_listed":1,"syntology":null},{"url":"/paper/deep-rts-a-game-environment-for-deep","slug":"deep-rts-a-game-environment-for-deep","title":"Deep RTS: A Game Environment for Deep Reinforcement Learning in Real-Time Strategy Games","date":"2018-08-15","arxiv_id":"1808.05032","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-deep-reinforcement-learning","slug":"an-efficient-deep-reinforcement-learning","title":"An Efficient Deep Reinforcement Learning Model for Urban Traffic Control","date":"2018-08-06","arxiv_id":"1808.01876","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-share-and-hide-intentions-using","slug":"learning-to-share-and-hide-intentions-using","title":"Learning to Share and Hide Intentions using Information Regularization","date":"2018-08-06","arxiv_id":"1808.02093","repositories_listed":1,"syntology":null},{"url":"/paper/recogym-a-reinforcement-learning-environment","slug":"recogym-a-reinforcement-learning-environment","title":"RecoGym: A Reinforcement Learning Environment for the problem of Product Recommendation in Online Advertising","date":"2018-08-02","arxiv_id":"1808.00720","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recogym-a-reinforcement-learning-environment#ran","syntology_url":"https://syntology.ai/paper/1808.00720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.00720"}},"official":{"repos":["criteo-research/reco-gym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distantly-supervised-ner-with-partial","slug":"distantly-supervised-ner-with-partial","title":"Distantly Supervised NER with Partial Annotation Learning and Reinforcement Learning","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-1","slug":"multi-agent-deep-reinforcement-learning-for-1","title":"Multi-Agent Deep Reinforcement Learning for Dynamic Power Allocation in Wireless Networks","date":"2018-08-01","arxiv_id":"1808.00490","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-generative-adversarial-imitation","slug":"multi-agent-generative-adversarial-imitation","title":"Multi-Agent Generative Adversarial Imitation Learning","date":"2018-07-26","arxiv_id":"1807.09936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-generative-adversarial-imitation#ran","syntology_url":"https://syntology.ai/paper/1807.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.09936"}},"official":null}},{"url":"/paper/backprop-q-generalized-backpropagation-for","slug":"backprop-q-generalized-backpropagation-for","title":"Backprop-Q: Generalized Backpropagation for Stochastic Computation Graphs","date":"2018-07-25","arxiv_id":"1807.09511","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-a-report","slug":"multi-agent-reinforcement-learning-a-report","title":"Multi-Agent Reinforcement Learning: A Report on Challenges and Approaches","date":"2018-07-25","arxiv_id":"1807.09427","repositories_listed":1,"syntology":null},{"url":"/paper/safe-option-critic-learning-safety-in-the","slug":"safe-option-critic-learning-safety-in-the","title":"Safe Option-Critic: Learning Safety in the Option-Critic Architecture","date":"2018-07-21","arxiv_id":"1807.08060","repositories_listed":1,"syntology":null},{"url":"/paper/learning-heuristics-for-automated-reasoning","slug":"learning-heuristics-for-automated-reasoning","title":"Learning Heuristics for Quantified Boolean Formulas through Deep Reinforcement Learning","date":"2018-07-20","arxiv_id":"1807.08058","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-for-zero","slug":"hierarchical-reinforcement-learning-for-zero","title":"Hierarchical Reinforcement Learning for Zero-shot Generalization with Subtask Dependencies","date":"2018-07-19","arxiv_id":"1807.07665","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/hierarchical-reinforcement-learning-for-zero#ran","syntology_url":"https://syntology.ai/paper/1807.07665","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.07665"}},"official":{"repos":["srsohn/subtask-graph-execution"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/backplay-man-muss-immer-umkehren","slug":"backplay-man-muss-immer-umkehren","title":"Backplay: \"Man muss immer umkehren\"","date":"2018-07-18","arxiv_id":"1807.06919","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-swarm-systems","slug":"deep-reinforcement-learning-for-swarm-systems","title":"Deep Reinforcement Learning for Swarm Systems","date":"2018-07-17","arxiv_id":"1807.06613","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-listen-read-and-follow-score","slug":"learning-to-listen-read-and-follow-score","title":"Learning to Listen, Read, and Follow: Score Following as a Reinforcement Learning Game","date":"2018-07-17","arxiv_id":"1807.06391","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-to-listen-read-and-follow-score#ran","syntology_url":"https://syntology.ai/paper/1807.06391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.06391"}},"official":{"repos":["CPJKU/score_following_game"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/is-q-learning-provably-efficient","slug":"is-q-learning-provably-efficient","title":"Is Q-learning Provably Efficient?","date":"2018-07-10","arxiv_id":"1807.03765","repositories_listed":1,"syntology":null},{"url":"/paper/financial-trading-as-a-game-a-deep","slug":"financial-trading-as-a-game-a-deep","title":"Financial Trading as a Game: A Deep Reinforcement Learning Approach","date":"2018-07-08","arxiv_id":"1807.02787","repositories_listed":1,"syntology":null},{"url":"/paper/learning-goal-oriented-visual-dialog-via","slug":"learning-goal-oriented-visual-dialog-via","title":"Learning Goal-Oriented Visual Dialog via Tempered Policy Gradient","date":"2018-07-02","arxiv_id":"1807.00737","repositories_listed":1,"syntology":null},{"url":"/paper/accurate-uncertainties-for-deep-learning","slug":"accurate-uncertainties-for-deep-learning","title":"Accurate Uncertainties for Deep Learning Using Calibrated Regression","date":"2018-07-01","arxiv_id":"1807.00263","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-continuous","slug":"deep-reinforcement-learning-in-continuous","title":"Deep Reinforcement Learning in Continuous Action Spaces: a Case Study in the Game of Simulated Curling","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dice-the-infinitely-differentiable-monte-1","slug":"dice-the-infinitely-differentiable-monte-1","title":"DiCE: The Infinitely Differentiable Monte Carlo Estimator","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-how-to-actively-learn-a-deep","slug":"learning-how-to-actively-learn-a-deep","title":"Learning How to Actively Learn: A Deep Imitation Learning Approach","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sequicity-simplifying-task-oriented-dialogue","slug":"sequicity-simplifying-task-oriented-dialogue","title":"Sequicity: Simplifying Task-oriented Dialogue Systems with Single Sequence-to-Sequence Architectures","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/universal-planning-networks-learning","slug":"universal-planning-networks-learning","title":"Universal Planning Networks: Learning Generalizable Representations for Visuomotor Control","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/using-reward-machines-for-high-level-task","slug":"using-reward-machines-for-high-level-task","title":"Using Reward Machines for High-Level Task Specification and Decomposition in Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/one-shot-learning-of-multi-step-tasks-from","slug":"one-shot-learning-of-multi-step-tasks-from","title":"One-Shot Learning of Multi-Step Tasks from Observation via Activity Localization in Auxiliary Video","date":"2018-06-29","arxiv_id":"1806.11244","repositories_listed":1,"syntology":null},{"url":"/paper/textworld-a-learning-environment-for-text","slug":"textworld-a-learning-environment-for-text","title":"TextWorld: A Learning Environment for Text-based Games","date":"2018-06-29","arxiv_id":"1806.11532","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textworld-a-learning-environment-for-text#ran","syntology_url":"https://syntology.ai/paper/1806.11532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.11532"}},"official":null}},{"url":"/paper/illuminating-generalization-in-deep","slug":"illuminating-generalization-in-deep","title":"Illuminating Generalization in Deep Reinforcement Learning through Procedural Level Generation","date":"2018-06-28","arxiv_id":"1806.10729","repositories_listed":1,"syntology":null},{"url":"/paper/qt-opt-scalable-deep-reinforcement-learning","slug":"qt-opt-scalable-deep-reinforcement-learning","title":"QT-Opt: Scalable Deep Reinforcement Learning for Vision-Based Robotic Manipulation","date":"2018-06-27","arxiv_id":"1806.10293","repositories_listed":1,"syntology":null},{"url":"/paper/guided-evolutionary-strategies-escaping-the","slug":"guided-evolutionary-strategies-escaping-the","title":"Guided evolutionary strategies: Augmenting random search with surrogate gradients","date":"2018-06-26","arxiv_id":"1806.10230","repositories_listed":1,"syntology":null},{"url":"/paper/a-tour-of-reinforcement-learning-the-view","slug":"a-tour-of-reinforcement-learning-the-view","title":"A Tour of Reinforcement Learning: The View from Continuous Control","date":"2018-06-25","arxiv_id":"1806.09460","repositories_listed":1,"syntology":null},{"url":"/paper/multi-objective-model-based-policy-search-for","slug":"multi-objective-model-based-policy-search-for","title":"Multi-objective Model-based Policy Search for Data-efficient Learning with Sparse Rewards","date":"2018-06-25","arxiv_id":"1806.09351","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-backprop-online-alternating","slug":"beyond-backprop-online-alternating","title":"Beyond Backprop: Online Alternating Minimization with Auxiliary Variables","date":"2018-06-24","arxiv_id":"1806.09077","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-backprop-online-alternating#ran","syntology_url":"https://syntology.ai/paper/1806.09077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.09077"}},"official":{"repos":["IBM/online-alt-min"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-surgical","slug":"deep-reinforcement-learning-for-surgical","title":"Deep Reinforcement Learning for Surgical Gesture Segmentation and Classification","date":"2018-06-21","arxiv_id":"1806.08089","repositories_listed":1,"syntology":null},{"url":"/paper/how-many-random-seeds-statistical-power","slug":"how-many-random-seeds-statistical-power","title":"How Many Random Seeds? Statistical Power Analysis in Deep Reinforcement Learning Experiments","date":"2018-06-21","arxiv_id":"1806.08295","repositories_listed":1,"syntology":null},{"url":"/paper/sim-to-real-reinforcement-learning-for","slug":"sim-to-real-reinforcement-learning-for","title":"Sim-to-Real Reinforcement Learning for Deformable Object Manipulation","date":"2018-06-20","arxiv_id":"1806.07851","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sim-to-real-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.07851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07851"}},"official":{"repos":["JanMatas/Rainbow_ddpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mlpack-3-a-fast-flexible-machine-learning","slug":"mlpack-3-a-fast-flexible-machine-learning","title":"mlpack 3: a fast, flexible machine learning library","date":"2018-06-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/barc-backward-reachability-curriculum-for","slug":"barc-backward-reachability-curriculum-for","title":"BaRC: Backward Reachability Curriculum for Robotic Reinforcement Learning","date":"2018-06-16","arxiv_id":"1806.06161","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/barc-backward-reachability-curriculum-for#ran","syntology_url":"https://syntology.ai/paper/1806.06161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.06161"}},"official":{"repos":["StanfordASL/BaRC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-image-data-preprocessing-with-deep","slug":"automated-image-data-preprocessing-with-deep","title":"Automated Image Data Preprocessing with Deep Reinforcement Learning","date":"2018-06-15","arxiv_id":"1806.05886","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-variance-reduced-policy-gradient","slug":"stochastic-variance-reduced-policy-gradient","title":"Stochastic Variance-Reduced Policy Gradient","date":"2018-06-14","arxiv_id":"1806.05618","repositories_listed":1,"syntology":null},{"url":"/paper/marginal-policy-gradients-a-unified-family-of","slug":"marginal-policy-gradients-a-unified-family-of","title":"Marginal Policy Gradients: A Unified Family of Estimators for Bounded Action Spaces with Applications","date":"2018-06-13","arxiv_id":"1806.05134","repositories_listed":1,"syntology":null},{"url":"/paper/the-potential-of-the-return-distribution-for","slug":"the-potential-of-the-return-distribution-for","title":"The Potential of the Return Distribution for Exploration in RL","date":"2018-06-11","arxiv_id":"1806.04242","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-potential-of-the-return-distribution-for#ran","syntology_url":"https://syntology.ai/paper/1806.04242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.04242"}},"official":{"repos":["tmoer/return_distribution_exploration"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-chinese-zero","slug":"deep-reinforcement-learning-for-chinese-zero","title":"Deep Reinforcement Learning for Chinese Zero pronoun Resolution","date":"2018-06-10","arxiv_id":"1806.03711","repositories_listed":1,"syntology":null},{"url":"/paper/conversational-recommender-system","slug":"conversational-recommender-system","title":"Conversational Recommender System","date":"2018-06-08","arxiv_id":"1806.03277","repositories_listed":1,"syntology":null},{"url":"/paper/randomized-prior-functions-for-deep","slug":"randomized-prior-functions-for-deep","title":"Randomized Prior Functions for Deep Reinforcement Learning","date":"2018-06-08","arxiv_id":"1806.03335","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/randomized-prior-functions-for-deep#ran","syntology_url":"https://syntology.ai/paper/1806.03335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.03335"}},"official":null}},{"url":"/paper/temporal-difference-variational-auto-encoder","slug":"temporal-difference-variational-auto-encoder","title":"Temporal Difference Variational Auto-Encoder","date":"2018-06-08","arxiv_id":"1806.03107","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-attack-on-graph-structured-data","slug":"adversarial-attack-on-graph-structured-data","title":"Adversarial Attack on Graph Structured Data","date":"2018-06-06","arxiv_id":"1806.02371","repositories_listed":1,"syntology":null},{"url":"/paper/deep-variational-reinforcement-learning-for","slug":"deep-variational-reinforcement-learning-for","title":"Deep Variational Reinforcement Learning for POMDPs","date":"2018-06-06","arxiv_id":"1806.02426","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-variational-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.02426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02426"}},"official":{"repos":["maximilianigl/DVRL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-understand-goal-specifications-by","slug":"learning-to-understand-goal-specifications-by","title":"Learning to Understand Goal Specifications by Modelling Reward","date":"2018-06-05","arxiv_id":"1806.01946","repositories_listed":1,"syntology":null},{"url":"/paper/tafe-net-task-aware-feature-embeddings-for","slug":"tafe-net-task-aware-feature-embeddings-for","title":"Deep Mixture of Experts via Shallow Embedding","date":"2018-06-05","arxiv_id":"1806.01531","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tafe-net-task-aware-feature-embeddings-for#ran","syntology_url":"https://syntology.ai/paper/1806.01531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01531"}},"official":null}},{"url":"/paper/bindsnet-a-machine-learning-oriented-spiking","slug":"bindsnet-a-machine-learning-oriented-spiking","title":"BindsNET: A machine learning-oriented spiking neural networks library in Python","date":"2018-06-04","arxiv_id":"1806.01423","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-in-high-dimensional-reinforcement","slug":"challenges-in-high-dimensional-reinforcement","title":"Challenges in High-dimensional Reinforcement Learning with Evolution Strategies","date":"2018-06-04","arxiv_id":"1806.01224","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/challenges-in-high-dimensional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1806.01224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01224"}},"official":{"repos":["NiMlr/High-Dim-ES-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/playing-atari-with-six-neurons","slug":"playing-atari-with-six-neurons","title":"Playing Atari with Six Neurons","date":"2018-06-04","arxiv_id":"1806.01363","repositories_listed":1,"syntology":null},{"url":"/paper/td-or-not-td-analyzing-the-role-of-temporal","slug":"td-or-not-td-analyzing-the-role-of-temporal","title":"TD or not TD: Analyzing the Role of Temporal Differencing in Deep Reinforcement Learning","date":"2018-06-04","arxiv_id":"1806.01175","repositories_listed":1,"syntology":null},{"url":"/paper/being-curious-about-the-answers-to-questions","slug":"being-curious-about-the-answers-to-questions","title":"Being curious about the answers to questions: novelty search with learned attention","date":"2018-06-01","arxiv_id":"1806.00201","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-of-region","slug":"deep-reinforcement-learning-of-region","title":"Deep Reinforcement Learning of Region Proposal Networks for Object Detection","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-continual-learning","slug":"reinforced-continual-learning","title":"Reinforced Continual Learning","date":"2018-05-31","arxiv_id":"1805.12369","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-2","slug":"sample-efficient-deep-reinforcement-learning-2","title":"Sample-Efficient Deep Reinforcement Learning via Episodic Backward Update","date":"2018-05-31","arxiv_id":"1805.12375","repositories_listed":1,"syntology":null},{"url":"/paper/playing-hard-exploration-games-by-watching","slug":"playing-hard-exploration-games-by-watching","title":"Playing hard exploration games by watching YouTube","date":"2018-05-29","arxiv_id":"1805.11592","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/playing-hard-exploration-games-by-watching#ran","syntology_url":"https://syntology.ai/paper/1805.11592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.11592"}},"official":null}},{"url":"/paper/supervised-policy-update-for-deep","slug":"supervised-policy-update-for-deep","title":"Supervised Policy Update for Deep Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11706","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-self-play","slug":"memory-augmented-self-play","title":"Memory Augmented Self-Play","date":"2018-05-28","arxiv_id":"1805.11016","repositories_listed":1,"syntology":null},{"url":"/paper/reward-constrained-policy-optimization","slug":"reward-constrained-policy-optimization","title":"Reward Constrained Policy Optimization","date":"2018-05-28","arxiv_id":"1805.11074","repositories_listed":1,"syntology":null},{"url":"/paper/reliability-and-learnability-of-human-bandit","slug":"reliability-and-learnability-of-human-bandit","title":"Reliability and Learnability of Human Bandit Feedback for Sequence-to-Sequence Reinforcement Learning","date":"2018-05-27","arxiv_id":"1805.10627","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reliability-and-learnability-of-human-bandit#ran","syntology_url":"https://syntology.ai/paper/1805.10627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.10627"}},"official":null}},{"url":"/paper/myopic-bayesian-design-of-experiments-via","slug":"myopic-bayesian-design-of-experiments-via","title":"Myopic Bayesian Design of Experiments via Posterior Sampling and Probabilistic Programming","date":"2018-05-25","arxiv_id":"1805.09964","repositories_listed":1,"syntology":null},{"url":"/paper/object-oriented-dynamics-predictor","slug":"object-oriented-dynamics-predictor","title":"Object-Oriented Dynamics Predictor","date":"2018-05-25","arxiv_id":"1806.07371","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-dual-machine-translation","slug":"zero-shot-dual-machine-translation","title":"Zero-Shot Dual Machine Translation","date":"2018-05-25","arxiv_id":"1805.10338","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-trainer-for-model-based","slug":"intelligent-trainer-for-model-based","title":"Intelligent Trainer for Model-Based Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09496","repositories_listed":1,"syntology":null},{"url":"/paper/meta-gradient-reinforcement-learning","slug":"meta-gradient-reinforcement-learning","title":"Meta-Gradient Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09801","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meta-gradient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1805.09801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09801"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-of-marked","slug":"deep-reinforcement-learning-of-marked","title":"Deep Reinforcement Learning of Marked Temporal Point Processes","date":"2018-05-23","arxiv_id":"1805.09360","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-coordinated-exploration-in","slug":"scalable-coordinated-exploration-in","title":"Scalable Coordinated Exploration in Concurrent Reinforcement Learning","date":"2018-05-23","arxiv_id":"1805.08948","repositories_listed":1,"syntology":null},{"url":"/paper/guided-feature-transformation-gft-a-neural","slug":"guided-feature-transformation-gft-a-neural","title":"Guided Feature Transformation (GFT): A Neural Language Grounding Module for Embodied Agents","date":"2018-05-22","arxiv_id":"1805.08329","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-maximum-entropy-inverse","slug":"multi-task-maximum-entropy-inverse","title":"Multi-task Maximum Entropy Inverse Reinforcement Learning","date":"2018-05-22","arxiv_id":"1805.08882","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-task-maximum-entropy-inverse#ran","syntology_url":"https://syntology.ai/paper/1805.08882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.08882"}},"official":{"repos":["HumanCompatibleAI/population-irl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/where-do-you-think-youre-going-inferring","slug":"where-do-you-think-youre-going-inferring","title":"Where Do You Think You're Going?: Inferring Beliefs about Dynamics from Behavior","date":"2018-05-21","arxiv_id":"1805.08010","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/where-do-you-think-youre-going-inferring#ran","syntology_url":"https://syntology.ai/paper/1805.08010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.08010"}},"official":{"repos":["rddy/isql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-lyapunov-based-approach-to-safe","slug":"a-lyapunov-based-approach-to-safe","title":"A Lyapunov-based Approach to Safe Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07708","repositories_listed":1,"syntology":null},{"url":"/paper/machine-teaching-for-inverse-reinforcement","slug":"machine-teaching-for-inverse-reinforcement","title":"Machine Teaching for Inverse Reinforcement Learning: Algorithms and Applications","date":"2018-05-20","arxiv_id":"1805.07687","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/machine-teaching-for-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1805.07687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07687"}},"official":{"repos":["dsbrown1331/machine-teaching-irl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-policy-learning-from-observations","slug":"safe-policy-learning-from-observations","title":"Constrained Policy Improvement for Safe and Efficient Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07805","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-video-object-segmentation-for","slug":"unsupervised-video-object-segmentation-for","title":"Unsupervised Video Object Segmentation for Deep Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07780","repositories_listed":1,"syntology":null},{"url":"/paper/improving-image-captioning-with-conditional","slug":"improving-image-captioning-with-conditional","title":"Improving Image Captioning with Conditional Generative Adversarial Nets","date":"2018-05-18","arxiv_id":"1805.07112","repositories_listed":1,"syntology":null},{"url":"/paper/learning-time-sensitive-strategies-in-space","slug":"learning-time-sensitive-strategies-in-space","title":"Learning Time-Sensitive Strategies in Space Fortress","date":"2018-05-17","arxiv_id":"1805.06824","repositories_listed":1,"syntology":null}],"record_sha256":"a677ced7573057cf4a188544cda12e85fef46759e5819cbd59b2b510a4305c5f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}