{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/15","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":59,"rows_per_page":100,"rows":[1401,1500],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/14","next":"/task/deep-reinforcement-learning/papers/16","papers":[{"url":"/paper/reinforcement-co-learning-of-deep-and-spiking","slug":"reinforcement-co-learning-of-deep-and-spiking","title":"Reinforcement co-Learning of Deep and Spiking Neural Networks for Energy-Efficient Mapless Navigation with Neuromorphic Hardware","date":"2020-03-02","arxiv_id":"2003.01157","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-co-learning-of-deep-and-spiking#ran","syntology_url":"https://syntology.ai/paper/2003.01157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.01157"}},"official":{"repos":["combra-lab/spiking-ddpg-mapless-navigation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-catastrophic-interference-in-atari-2600","slug":"on-catastrophic-interference-in-atari-2600","title":"On Catastrophic Interference in Atari 2600 Games","date":"2020-02-28","arxiv_id":"2002.12499","repositories_listed":1,"syntology":null},{"url":"/paper/gamma-reward-a-novel-multi-agent","slug":"gamma-reward-a-novel-multi-agent","title":"Learning Scalable Multi-Agent Coordination by Spatial Differentiation for Traffic Signal Control","date":"2020-02-27","arxiv_id":"2002.11874","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-deep-reinforcement-learning-with","slug":"off-policy-deep-reinforcement-learning-with","title":"Off-Policy Deep Reinforcement Learning with Analogous Disentangled Exploration","date":"2020-02-25","arxiv_id":"2002.10738","repositories_listed":1,"syntology":null},{"url":"/paper/reconfigurable-intelligent-surface-assisted","slug":"reconfigurable-intelligent-surface-assisted","title":"Reconfigurable Intelligent Surface Assisted Multiuser MISO Systems Exploiting Deep Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10072","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-particle-filter-reinforcement-1","slug":"discriminative-particle-filter-reinforcement-1","title":"Discriminative Particle Filter Reinforcement Learning for Complex Partial Observations","date":"2020-02-23","arxiv_id":"2002.09884","repositories_listed":1,"syntology":null},{"url":"/paper/tuning-free-plug-and-play-proximal-algorithm","slug":"tuning-free-plug-and-play-proximal-algorithm","title":"Tuning-free Plug-and-Play Proximal Algorithm for Inverse Imaging Problems","date":"2020-02-22","arxiv_id":"2002.09611","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-walk-in-the-real-world-with","slug":"learning-to-walk-in-the-real-world-with","title":"Learning to Walk in the Real World with Minimal Human Effort","date":"2020-02-20","arxiv_id":"2002.08550","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-through","slug":"efficient-deep-reinforcement-learning-through","title":"Efficient Deep Reinforcement Learning via Adaptive Policy Transfer","date":"2020-02-19","arxiv_id":"2002.08037","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-molecular-design","slug":"reinforcement-learning-for-molecular-design","title":"Reinforcement Learning for Molecular Design Guided by Quantum Mechanics","date":"2020-02-18","arxiv_id":"2002.07717","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-active-learning-for-image-1","slug":"reinforced-active-learning-for-image-1","title":"Reinforced active learning for image segmentation","date":"2020-02-16","arxiv_id":"2002.06583","repositories_listed":1,"syntology":null},{"url":"/paper/deep-rl-agent-for-a-real-time-action-strategy","slug":"deep-rl-agent-for-a-real-time-action-strategy","title":"Deep RL Agent for a Real-Time Action Strategy Game","date":"2020-02-15","arxiv_id":"2002.06290","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/integrating-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.00434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00434"}},"official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/challenges-and-countermeasures-for","slug":"challenges-and-countermeasures-for","title":"Challenges and Countermeasures for Adversarial Attacks on Deep Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09684","repositories_listed":1,"syntology":null},{"url":"/paper/rotation-translation-and-cropping-for-zero","slug":"rotation-translation-and-cropping-for-zero","title":"Rotation, Translation, and Cropping for Zero-Shot Generalization","date":"2020-01-27","arxiv_id":"2001.09908","repositories_listed":1,"syntology":null},{"url":"/paper/sarl-deep-reinforcement-learning-based-human","slug":"sarl-deep-reinforcement-learning-based-human","title":"SARL*: Deep Reinforcement Learning based Human-Aware Navigation for Mobile Robot in Indoor Environments","date":"2020-01-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pops-policy-pruning-and-shrinking-for-deep","slug":"pops-policy-pruning-and-shrinking-for-deep","title":"PoPS: Policy Pruning and Shrinking for Deep Reinforcement Learning","date":"2020-01-14","arxiv_id":"2001.05012","repositories_listed":1,"syntology":null},{"url":"/paper/closed-loop-deep-learning-generating-forward","slug":"closed-loop-deep-learning-generating-forward","title":"Closed-loop deep learning: generating forward models with back-propagation","date":"2020-01-09","arxiv_id":"2001.02970","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-human","slug":"deep-reinforcement-learning-for-active-human","title":"Deep Reinforcement Learning for Active Human Pose Estimation","date":"2020-01-07","arxiv_id":"2001.02024","repositories_listed":1,"syntology":null},{"url":"/paper/an-optimistic-perspective-on-offline-deep","slug":"an-optimistic-perspective-on-offline-deep","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-deep-neuroevolution-via-deep","slug":"improving-deep-neuroevolution-via-deep","title":"Deep Innovation Protection: Confronting the Credit Assignment Problem in Training Heterogeneous Neural Architectures","date":"2019-12-29","arxiv_id":"2001.01683","repositories_listed":1,"syntology":null},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-variable-ordering-heuristics-for","slug":"learning-variable-ordering-heuristics-for","title":"Learning Variable Ordering Heuristics for Solving Constraint Satisfaction Problems","date":"2019-12-23","arxiv_id":"1912.10762","repositories_listed":1,"syntology":null},{"url":"/paper/variational-recurrent-models-for-solving-1","slug":"variational-recurrent-models-for-solving-1","title":"Variational Recurrent Models for Solving Partially Observable Control Tasks","date":"2019-12-23","arxiv_id":"1912.10703","repositories_listed":1,"syntology":null},{"url":"/paper/pixelrl-fully-convolutional-network-with","slug":"pixelrl-fully-convolutional-network-with","title":"PixelRL: Fully Convolutional Network with Reinforcement Learning for Image Processing","date":"2019-12-16","arxiv_id":"1912.07190","repositories_listed":1,"syntology":null},{"url":"/paper/dota-2-with-large-scale-deep-reinforcement","slug":"dota-2-with-large-scale-deep-reinforcement","title":"Dota 2 with Large Scale Deep Reinforcement Learning","date":"2019-12-13","arxiv_id":"1912.06680","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dota-2-with-large-scale-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1912.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.06680"}},"official":null}},{"url":"/paper/learning-improvement-heuristics-for-solving","slug":"learning-improvement-heuristics-for-solving","title":"Learning Improvement Heuristics for Solving Routing Problems","date":"2019-12-12","arxiv_id":"1912.05784","repositories_listed":1,"syntology":null},{"url":"/paper/chainerrl-a-deep-reinforcement-learning","slug":"chainerrl-a-deep-reinforcement-learning","title":"ChainerRL: A Deep Reinforcement Learning Library","date":"2019-12-09","arxiv_id":"1912.03905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chainerrl-a-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1912.03905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03905"}},"official":{"repos":["chainer/chainerrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploratory-not-explanatory-counterfactual-1","slug":"exploratory-not-explanatory-counterfactual-1","title":"Exploratory Not Explanatory: Counterfactual Analysis of Saliency Maps for Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.05743","repositories_listed":1,"syntology":null},{"url":"/paper/increasing-performance-of-electric-vehicles","slug":"increasing-performance-of-electric-vehicles","title":"Increasing performance of electric vehicles in ride-hailing services using deep reinforcement learning","date":"2019-12-07","arxiv_id":"1912.03408","repositories_listed":1,"syntology":null},{"url":"/paper/valan-vision-and-language-agent-navigation","slug":"valan-vision-and-language-agent-navigation","title":"VALAN: Vision and Language Agent Navigation","date":"2019-12-06","arxiv_id":"1912.03241","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-guided-hindsight-experience-replay","slug":"curriculum-guided-hindsight-experience-replay","title":"Curriculum-guided Hindsight Experience Replay","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/domes-to-drones-self-supervised-active","slug":"domes-to-drones-self-supervised-active","title":"Domes to Drones: Self-Supervised Active Triangulation for 3D Human Pose Reconstruction","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-local-search-heuristics-for-boolean","slug":"learning-local-search-heuristics-for-boolean","title":"Learning Local Search Heuristics for Boolean Satisfiability","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reconciling-returns-with-experience-replay","slug":"reconciling-returns-with-experience-replay","title":"Reconciling λ-Returns with Experience Replay","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-similarity-graphs-constructed-by-deep","slug":"towards-similarity-graphs-constructed-by-deep","title":"Towards Similarity Graphs Constructed by Deep Reinforcement Learning","date":"2019-11-27","arxiv_id":"1911.12122","repositories_listed":1,"syntology":null},{"url":"/paper/join-query-optimization-with-deep","slug":"join-query-optimization-with-deep","title":"Join Query Optimization with Deep Reinforcement Learning Algorithms","date":"2019-11-26","arxiv_id":"1911.11689","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-control-of-a-fiber-manufacturing","slug":"dynamic-control-of-a-fiber-manufacturing","title":"Dynamic Control of a Fiber Manufacturing Process using Deep Reinforcement Learning","date":"2019-11-23","arxiv_id":"1911.10286","repositories_listed":1,"syntology":null},{"url":"/paper/deepsynth-program-synthesis-for-automatic","slug":"deepsynth-program-synthesis-for-automatic","title":"DeepSynth: Automata Synthesis for Automatic Task Segmentation in Deep Reinforcement Learning","date":"2019-11-22","arxiv_id":"1911.10244","repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-episodic-control","slug":"memory-efficient-episodic-control","title":"Memory-Efficient Episodic Control Reinforcement Learning with Dynamic Online k-means","date":"2019-11-21","arxiv_id":"1911.09560","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-explicitly","slug":"deep-reinforcement-learning-with-explicitly","title":"Classification with Costly Features in Hierarchical Deep Sets","date":"2019-11-20","arxiv_id":"1911.08756","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-task-agnostic-exploration-for","slug":"evaluating-task-agnostic-exploration-for","title":"Evaluating task-agnostic exploration for fixed-batch learning of arbitrary future tasks","date":"2019-11-20","arxiv_id":"1911.08666","repositories_listed":1,"syntology":null},{"url":"/paper/neural-approximate-dynamic-programming-for-on","slug":"neural-approximate-dynamic-programming-for-on","title":"Neural Approximate Dynamic Programming for On-Demand Ride-Pooling","date":"2019-11-20","arxiv_id":"1911.08842","repositories_listed":1,"syntology":null},{"url":"/paper/deep-tile-coder-an-efficient-sparse","slug":"deep-tile-coder-an-efficient-sparse","title":"Fuzzy Tiling Activations: A Simple Approach to Learning Sparse Representations Online","date":"2019-11-19","arxiv_id":"1911.08068","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-resource-allocation-in-stream","slug":"generalizable-resource-allocation-in-stream","title":"Generalizable Resource Allocation in Stream Processing via Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08517","repositories_listed":1,"syntology":null},{"url":"/paper/influence-aware-memory-for-deep-reinforcement-1","slug":"influence-aware-memory-for-deep-reinforcement-1","title":"Influence-aware Memory Architectures for Deep Reinforcement Learning","date":"2019-11-18","arxiv_id":"1911.07643","repositories_listed":1,"syntology":null},{"url":"/paper/drills-deep-reinforcement-learning-for-logic","slug":"drills-deep-reinforcement-learning-for-logic","title":"DRiLLS: Deep Reinforcement Learning for Logic Synthesis","date":"2019-11-11","arxiv_id":"1911.04021","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drills-deep-reinforcement-learning-for-logic#ran","syntology_url":"https://syntology.ai/paper/1911.04021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04021"}},"official":{"repos":["scale-lab/DRiLLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-connected-autonomous-driving","slug":"multi-agent-connected-autonomous-driving","title":"Multi-Agent Connected Autonomous Driving using Deep Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04175","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-connected-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/1911.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04175"}},"official":{"repos":["praveen-palanisamy/macad-gym"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-deep-reinforcement-learning-based-approach","slug":"a-deep-reinforcement-learning-based-approach","title":"A Deep Reinforcement Learning Approach to First-Order Logic Theorem Proving","date":"2019-11-05","arxiv_id":"1911.02065","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-numeric-concepts-in-multi-agent","slug":"emergence-of-numeric-concepts-in-multi-agent","title":"Emergence of Numeric Concepts in Multi-Agent Autonomous Communication","date":"2019-11-04","arxiv_id":"1911.01098","repositories_listed":1,"syntology":null},{"url":"/paper/cascaded-lstms-based-deep-reinforcement","slug":"cascaded-lstms-based-deep-reinforcement","title":"Cascaded LSTMs based Deep Reinforcement Learning for Goal-driven Dialogue","date":"2019-10-31","arxiv_id":"1910.14229","repositories_listed":1,"syntology":null},{"url":"/paper/191013406","slug":"191013406","title":"Generalization of Reinforcement Learners with Working and Episodic Memory","date":"2019-10-29","arxiv_id":"1910.13406","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-deep-energy-based","slug":"a-framework-for-deep-energy-based","title":"Quantum enhancements for deep reinforcement learning in large spaces","date":"2019-10-28","arxiv_id":"1910.12760","repositories_listed":1,"syntology":null},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-hol4","slug":"deep-reinforcement-learning-in-hol4","title":"Deep Reinforcement Learning for Synthesizing Functions in Higher-Order Logic","date":"2019-10-25","arxiv_id":"1910.11797","repositories_listed":1,"syntology":null},{"url":"/paper/learning-hierarchical-control-for-robust-in","slug":"learning-hierarchical-control-for-robust-in","title":"Learning Hierarchical Control for Robust In-Hand Manipulation","date":"2019-10-24","arxiv_id":"1910.10985","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-curiosity-driven-exploration","slug":"attention-based-curiosity-driven-exploration","title":"Attention-based Curiosity-driven Exploration in Deep Reinforcement Learning","date":"2019-10-23","arxiv_id":"1910.10840","repositories_listed":1,"syntology":null},{"url":"/paper/learning-humanoid-robot-running-skills","slug":"learning-humanoid-robot-running-skills","title":"Learning Humanoid Robot Running Skills through Proximal Policy Optimization","date":"2019-10-22","arxiv_id":"1910.10620","repositories_listed":1,"syntology":null},{"url":"/paper/dealing-with-sparse-rewards-in-reinforcement","slug":"dealing-with-sparse-rewards-in-reinforcement","title":"Dealing with Sparse Rewards in Reinforcement Learning","date":"2019-10-21","arxiv_id":"1910.09281","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-control-of","slug":"deep-reinforcement-learning-control-of","title":"Deep Reinforcement Learning Control of Quantum Cartpoles","date":"2019-10-21","arxiv_id":"1910.09200","repositories_listed":1,"syntology":null},{"url":"/paper/towards-more-sample-efficiency","slug":"towards-more-sample-efficiency","title":"Towards More Sample Efficiency in Reinforcement Learning with Data Augmentation","date":"2019-10-19","arxiv_id":"1910.09959","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-data-augmentation-by-learning-the","slug":"automatic-data-augmentation-by-learning-the","title":"Automatic Data Augmentation by Learning the Deterministic Policy","date":"2019-10-18","arxiv_id":"1910.08343","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-meets-graph","slug":"deep-reinforcement-learning-meets-graph","title":"Deep Reinforcement Learning meets Graph Neural Networks: exploring a routing optimization use case","date":"2019-10-16","arxiv_id":"1910.07421","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-meets-graph#ran","syntology_url":"https://syntology.ai/paper/1910.07421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07421"}},"official":{"repos":["knowledgedefinednetworking/DRL-GNN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstrapping-the-expressivity-with-model","slug":"bootstrapping-the-expressivity-with-model","title":"On the Expressivity of Neural Networks for Deep Reinforcement Learning","date":"2019-10-14","arxiv_id":"1910.05927","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-navigation-via-deep-reinforcement","slug":"autonomous-navigation-via-deep-reinforcement","title":"Autonomous Navigation via Deep Reinforcement Learning for Resource Constraint Edge Nodes using Transfer Learning","date":"2019-10-12","arxiv_id":"1910.05547","repositories_listed":1,"syntology":null},{"url":"/paper/from-visual-place-recognition-to-navigation","slug":"from-visual-place-recognition-to-navigation","title":"CityLearn: Diverse Real-World Environments for Sample-Efficient Navigation Policy Learning","date":"2019-10-10","arxiv_id":"1910.04335","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-network-for-angry-birds","slug":"deep-q-network-for-angry-birds","title":"Deep Q-Network for Angry Birds","date":"2019-10-04","arxiv_id":"1910.01806","repositories_listed":1,"syntology":null},{"url":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantized-reinforcement-learning-quarl#ran","syntology_url":"https://syntology.ai/paper/1910.01055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01055"}},"official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-evaluate-machine-learning-approaches","slug":"how-to-evaluate-machine-learning-approaches","title":"How to Evaluate Machine Learning Approaches for Combinatorial Optimization: Application to the Travelling Salesman Problem","date":"2019-09-28","arxiv_id":"1909.13121","repositories_listed":1,"syntology":null},{"url":"/paper/relational-graph-learning-for-crowd","slug":"relational-graph-learning-for-crowd","title":"Relational Graph Learning for Crowd Navigation","date":"2019-09-28","arxiv_id":"1909.13165","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-structures-for-value-based","slug":"harnessing-structures-for-value-based","title":"Harnessing Structures for Value-Based Planning and Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12255","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-structures-for-value-based#ran","syntology_url":"https://syntology.ai/paper/1909.12255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12255"}},"official":{"repos":["YyzHarry/SV-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy","slug":"v-mpo-on-policy-maximum-a-posteriori-policy","title":"V-MPO: On-Policy Maximum a Posteriori Policy Optimization for Discrete and Continuous Control","date":"2019-09-26","arxiv_id":"1909.12238","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy#ran","syntology_url":"https://syntology.ai/paper/1909.12238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12238"}},"official":null}},{"url":"/paper/learning-to-seek-autonomous-source-seeking","slug":"learning-to-seek-autonomous-source-seeking","title":"Learning to Seek: Autonomous Source Seeking with Deep Reinforcement Learning Onboard a Nano Drone Microcontroller","date":"2019-09-25","arxiv_id":"1909.11236","repositories_listed":1,"syntology":null},{"url":"/paper/invariant-transform-experience-replay","slug":"invariant-transform-experience-replay","title":"Invariant Transform Experience Replay: Data Augmentation for Deep Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.10707","repositories_listed":1,"syntology":null},{"url":"/paper/190910008","slug":"190910008","title":"Multi-task Learning and Catastrophic Forgetting in Continual Reinforcement Learning","date":"2019-09-22","arxiv_id":"1909.10008","repositories_listed":1,"syntology":null},{"url":"/paper/190909902","slug":"190909902","title":"Deep Reinforcement Learning with Modulated Hebbian plus Q Network Architecture","date":"2019-09-21","arxiv_id":"1909.09902","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-optimization-for-iterative-learning","slug":"bayesian-optimization-for-iterative-learning","title":"Bayesian Optimization for Iterative Learning","date":"2019-09-20","arxiv_id":"1909.09593","repositories_listed":1,"syntology":null},{"url":"/paper/selfie-drone-stick-a-natural-interface-for","slug":"selfie-drone-stick-a-natural-interface-for","title":"Selfie Drone Stick: A Natural Interface for Quadcopter Photography","date":"2019-09-14","arxiv_id":"1909.06491","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-reproducibility-by-evaluating","slug":"a-survey-on-reproducibility-by-evaluating","title":"A Survey on Reproducibility by Evaluating Deep Reinforcement Learning Algorithms on Real-World Robots","date":"2019-09-09","arxiv_id":"1909.03772","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-survey-on-reproducibility-by-evaluating#ran","syntology_url":"https://syntology.ai/paper/1909.03772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.03772"}},"official":{"repos":["dti-research/SenseActExperiments"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ac-teach-a-bayesian-actor-critic-method-for","slug":"ac-teach-a-bayesian-actor-critic-method-for","title":"AC-Teach: A Bayesian Actor-Critic Method for Policy Learning with an Ensemble of Suboptimal Teachers","date":"2019-09-09","arxiv_id":"1909.04121","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/ac-teach-a-bayesian-actor-critic-method-for#ran","syntology_url":"https://syntology.ai/paper/1909.04121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04121"}},"official":null}},{"url":"/paper/adversarial-policy-gradient-for-deep-learning","slug":"adversarial-policy-gradient-for-deep-learning","title":"Adversarial Policy Gradient for Deep Learning Image Augmentation","date":"2019-09-09","arxiv_id":"1909.04108","repositories_listed":1,"syntology":null},{"url":"/paper/regularized-anderson-acceleration-for-off","slug":"regularized-anderson-acceleration-for-off","title":"Regularized Anderson Acceleration for Off-Policy Deep Reinforcement Learning","date":"2019-09-07","arxiv_id":"1909.03245","repositories_listed":1,"syntology":null},{"url":"/paper/drlviz-understanding-decisions-and-memory-in","slug":"drlviz-understanding-decisions-and-memory-in","title":"DRLViz: Understanding Decisions and Memory in Deep Reinforcement Learning","date":"2019-09-06","arxiv_id":"1909.02982","repositories_listed":1,"syntology":null},{"url":"/paper/spatiotemporally-constrained-action-space","slug":"spatiotemporally-constrained-action-space","title":"Spatiotemporally Constrained Action Space Attacks on Deep Reinforcement Learning Agents","date":"2019-09-05","arxiv_id":"1909.02583","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-language-learning-by-question","slug":"interactive-language-learning-by-question","title":"Interactive Language Learning by Question Answering","date":"2019-08-28","arxiv_id":"1908.10909","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interactive-language-learning-by-question#ran","syntology_url":"https://syntology.ai/paper/1908.10909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10909"}},"official":{"repos":["xingdi-eric-yuan/qait_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-text-classification-with","slug":"hierarchical-text-classification-with","title":"Hierarchical Text Classification with Reinforced Label Assignment","date":"2019-08-27","arxiv_id":"1908.10419","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-world-earth","slug":"deep-reinforcement-learning-in-world-earth","title":"Deep reinforcement learning in World-Earth system models to discover sustainable management strategies","date":"2019-08-15","arxiv_id":"1908.05567","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-in-world-earth#ran","syntology_url":"https://syntology.ai/paper/1908.05567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.05567"}},"official":{"repos":["fstrnad/pyDRLinWESM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-question-refinement-with-deep","slug":"generative-question-refinement-with-deep","title":"Generative Question Refinement with Deep Reinforcement Learning in Retrieval-based QA System","date":"2019-08-13","arxiv_id":"1908.05604","repositories_listed":1,"syntology":null},{"url":"/paper/is-deep-reinforcement-learning-really","slug":"is-deep-reinforcement-learning-really","title":"Is Deep Reinforcement Learning Really Superhuman on Atari? Leveling the playing field","date":"2019-08-13","arxiv_id":"1908.04683","repositories_listed":1,"syntology":null},{"url":"/paper/a-review-on-deep-reinforcement-learning-for","slug":"a-review-on-deep-reinforcement-learning-for","title":"A review on Deep Reinforcement Learning for Fluid Mechanics","date":"2019-08-12","arxiv_id":"1908.04127","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-navigation-using-deep","slug":"vision-based-navigation-using-deep","title":"Vision-based Navigation Using Deep Reinforcement Learning","date":"2019-08-08","arxiv_id":"1908.03627","repositories_listed":1,"syntology":null},{"url":"/paper/free-lunch-saliency-via-attention-in-atari","slug":"free-lunch-saliency-via-attention-in-atari","title":"Free-Lunch Saliency via Attention in Atari Agents","date":"2019-08-07","arxiv_id":"1908.02511","repositories_listed":1,"syntology":null},{"url":"/paper/minerl-a-large-scale-dataset-of-minecraft","slug":"minerl-a-large-scale-dataset-of-minecraft","title":"MineRL: A Large-Scale Dataset of Minecraft Demonstrations","date":"2019-07-29","arxiv_id":"1907.13440","repositories_listed":1,"syntology":null},{"url":"/paper/making-sense-of-vision-and-touch-learning","slug":"making-sense-of-vision-and-touch-learning","title":"Making Sense of Vision and Touch: Learning Multimodal Representations for Contact-Rich Tasks","date":"2019-07-28","arxiv_id":"1907.13098","repositories_listed":1,"syntology":null},{"url":"/paper/towards-model-based-reinforcement-learning","slug":"towards-model-based-reinforcement-learning","title":"Towards Model-based Reinforcement Learning for Industry-near Environments","date":"2019-07-27","arxiv_id":"1907.11971","repositories_listed":1,"syntology":null},{"url":"/paper/action-semantics-network-considering-the","slug":"action-semantics-network-considering-the","title":"Action Semantics Network: Considering the Effects of Actions in Multiagent Systems","date":"2019-07-26","arxiv_id":"1907.11461","repositories_listed":1,"syntology":null},{"url":"/paper/muscle-actuated-human-simulation-and-control","slug":"muscle-actuated-human-simulation-and-control","title":"Muscle-actuated Human Simulation and Control","date":"2019-07-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/characterizing-attacks-on-deep-reinforcement","slug":"characterizing-attacks-on-deep-reinforcement","title":"Characterizing Attacks on Deep Reinforcement Learning","date":"2019-07-21","arxiv_id":"1907.09470","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/characterizing-attacks-on-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.09470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.09470"}},"official":null}},{"url":"/paper/ppo-dash-improving-generalization-in-deep","slug":"ppo-dash-improving-generalization-in-deep","title":"PPO Dash: Improving Generalization in Deep Reinforcement Learning","date":"2019-07-15","arxiv_id":"1907.06704","repositories_listed":1,"syntology":null}],"record_sha256":"1109d12d54f51520d26fd042e1ce0b8a654dc306e2ffcc1b9c4e9540f33139d0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}