{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/16","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":16,"pages_in_order":59,"rows_per_page":100,"rows":[1501,1600],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/15","next":"/task/deep-reinforcement-learning/papers/17","papers":[{"url":"/paper/proximal-policy-optimization-with-mixed","slug":"proximal-policy-optimization-with-mixed","title":"Proximal Policy Optimization with Mixed Distributed Training","date":"2019-07-15","arxiv_id":"1907.06479","repositories_listed":1,"syntology":null},{"url":"/paper/deep-lagrangian-networks-for-end-to-end","slug":"deep-lagrangian-networks-for-end-to-end","title":"Deep Lagrangian Networks for end-to-end learning of energy-based control for under-actuated systems","date":"2019-07-10","arxiv_id":"1907.04489","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-multi-task-deep-reinforcement","slug":"attentive-multi-task-deep-reinforcement","title":"Attentive Multi-Task Deep Reinforcement Learning","date":"2019-07-05","arxiv_id":"1907.02874","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/attentive-multi-task-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.02874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.02874"}},"official":{"repos":["braemt/attentive-multi-task-deep-reinforcement-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-quantum-circuits-and-deep","slug":"variational-quantum-circuits-and-deep","title":"Variational Quantum Circuits for Deep Reinforcement Learning","date":"2019-06-30","arxiv_id":"1907.00397","repositories_listed":1,"syntology":null},{"url":"/paper/way-off-policy-batch-deep-reinforcement","slug":"way-off-policy-batch-deep-reinforcement","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","date":"2019-06-30","arxiv_id":"1907.00456","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/way-off-policy-batch-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.00456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.00456"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-deep-reinforcement-learning-approach-for-1","slug":"a-deep-reinforcement-learning-approach-for-1","title":"A Deep Reinforcement Learning Approach for Global Routing","date":"2019-06-20","arxiv_id":"1906.08809","repositories_listed":1,"syntology":null},{"url":"/paper/a-hierarchical-architecture-for-sequential","slug":"a-hierarchical-architecture-for-sequential","title":"A Hierarchical Architecture for Sequential Decision-Making in Autonomous Driving using Deep Reinforcement Learning","date":"2019-06-20","arxiv_id":"1906.08464","repositories_listed":1,"syntology":null},{"url":"/paper/calibrated-model-based-deep-reinforcement","slug":"calibrated-model-based-deep-reinforcement","title":"Calibrated Model-Based Deep Reinforcement Learning","date":"2019-06-19","arxiv_id":"1906.08312","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-industrial","slug":"deep-reinforcement-learning-for-industrial","title":"Deep Reinforcement Learning for Industrial Insertion Tasks with Visual Inputs and Natural Rewards","date":"2019-06-13","arxiv_id":"1906.05841","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-for-industrial#ran","syntology_url":"https://syntology.ai/paper/1906.05841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05841"}},"official":null}},{"url":"/paper/gossip-based-actor-learner-architectures-for","slug":"gossip-based-actor-learner-architectures-for","title":"Gossip-based Actor-Learner Architectures for Deep Reinforcement Learning","date":"2019-06-09","arxiv_id":"1906.04585","repositories_listed":1,"syntology":null},{"url":"/paper/improving-exploration-in-soft-actor-critic","slug":"improving-exploration-in-soft-actor-critic","title":"Improving Exploration in Soft-Actor-Critic with Normalizing Flows Policies","date":"2019-06-06","arxiv_id":"1906.02771","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-exploration-in-soft-actor-critic#ran","syntology_url":"https://syntology.ai/paper/1906.02771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.02771"}},"official":{"repos":["joeybose/FloRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-with-low-complexity#ran","syntology_url":"https://syntology.ai/paper/1906.01695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01695"}},"official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/190600421","slug":"190600421","title":"Air Learning: A Deep Reinforcement Learning Gym for Autonomous Aerial Robot Visual Navigation","date":"2019-06-02","arxiv_id":"1906.00421","repositories_listed":1,"syntology":null},{"url":"/paper/190600190","slug":"190600190","title":"Neural Replicator Dynamics","date":"2019-06-01","arxiv_id":"1906.00190","repositories_listed":1,"syntology":null},{"url":"/paper/interval-timing-in-deep-reinforcement","slug":"interval-timing-in-deep-reinforcement","title":"Interval timing in deep reinforcement learning agents","date":"2019-05-31","arxiv_id":"1905.13469","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-modeling-of-temporal-credit","slug":"sequence-modeling-of-temporal-credit","title":"Sequence Modeling of Temporal Credit Assignment for Episodic Reinforcement Learning","date":"2019-05-31","arxiv_id":"1905.13420","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequence-modeling-of-temporal-credit#ran","syntology_url":"https://syntology.ai/paper/1905.13420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13420"}},"official":null}},{"url":"/paper/snooping-attacks-on-deep-reinforcement","slug":"snooping-attacks-on-deep-reinforcement","title":"Snooping Attacks on Deep Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.11832","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-discretize-solving-1d-scalar","slug":"learning-to-discretize-solving-1d-scalar","title":"Learning to Discretize: Solving 1D Scalar Conservation Laws via Deep Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11079","repositories_listed":1,"syntology":null},{"url":"/paper/neural-temporal-difference-learning-converges","slug":"neural-temporal-difference-learning-converges","title":"Neural Temporal-Difference and Q-Learning Provably Converge to Global Optima","date":"2019-05-24","arxiv_id":"1905.10027","repositories_listed":1,"syntology":null},{"url":"/paper/multi-hop-reading-comprehension-via-deep","slug":"multi-hop-reading-comprehension-via-deep","title":"Multi-hop Reading Comprehension via Deep Reinforcement Learning based Document Traversal","date":"2019-05-23","arxiv_id":"1905.09438","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-parameter","slug":"deep-reinforcement-learning-based-parameter","title":"Deep Reinforcement Learning Based Parameter Control in Differential Evolution","date":"2019-05-20","arxiv_id":"1905.08006","repositories_listed":1,"syntology":null},{"url":"/paper/task-agnostic-dynamics-priors-for-deep","slug":"task-agnostic-dynamics-priors-for-deep","title":"Task-Agnostic Dynamics Priors for Deep Reinforcement Learning","date":"2019-05-13","arxiv_id":"1905.04819","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-agnostic-dynamics-priors-for-deep#ran","syntology_url":"https://syntology.ai/paper/1905.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04819"}},"official":{"repos":["yilundu/task_agnostic_dynamics_prior"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/autonomous-management-of-energy-harvesting","slug":"autonomous-management-of-energy-harvesting","title":"Autonomous Management of Energy-Harvesting IoT Nodes Using Deep Reinforcement Learning","date":"2019-05-10","arxiv_id":"1905.04181","repositories_listed":1,"syntology":null},{"url":"/paper/190503389","slug":"190503389","title":"Learning to Evolve","date":"2019-05-08","arxiv_id":"1905.03389","repositories_listed":1,"syntology":null},{"url":"/paper/knowing-the-what-but-not-the-where-in","slug":"knowing-the-what-but-not-the-where-in","title":"Knowing The What But Not The Where in Bayesian Optimization","date":"2019-05-07","arxiv_id":"1905.02685","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowing-the-what-but-not-the-where-in#ran","syntology_url":"https://syntology.ai/paper/1905.02685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.02685"}},"official":{"repos":["ntienvu/KnownOptimum_BO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-ordinal-reinforcement-learning","slug":"deep-ordinal-reinforcement-learning","title":"Deep Ordinal Reinforcement Learning","date":"2019-05-06","arxiv_id":"1905.02005","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1905.00976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00976"}},"official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-video-game-to-real-robot-the-transfer","slug":"from-video-game-to-real-robot-the-transfer","title":"From Video Game to Real Robot: The Transfer between Action Spaces","date":"2019-05-02","arxiv_id":"1905.00741","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-policy-update","slug":"supervised-policy-update","title":"SUPERVISED POLICY UPDATE","date":"2019-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-logic-reinforcement-learning","slug":"neural-logic-reinforcement-learning","title":"Neural Logic Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.10729","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neural-logic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1904.10729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10729"}},"official":{"repos":["ZhengyaoJiang/NLRL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-minerl-competition-on-sample-efficient","slug":"the-minerl-competition-on-sample-efficient","title":"The MineRL 2019 Competition on Sample Efficient Reinforcement Learning using Human Priors","date":"2019-04-22","arxiv_id":"1904.10079","repositories_listed":1,"syntology":null},{"url":"/paper/190501360","slug":"190501360","title":"Skynet: A Top Deep RL Agent in the Inaugural Pommerman Team Competition","date":"2019-04-20","arxiv_id":"1905.01360","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-compositional-language-with-deep","slug":"emergence-of-compositional-language-with-deep","title":"Emergence of Compositional Language with Deep Generational Transmission","date":"2019-04-19","arxiv_id":"1904.09067","repositories_listed":1,"syntology":null},{"url":"/paper/towards-robust-deep-reinforcement-learning","slug":"towards-robust-deep-reinforcement-learning","title":"Towards Robust Deep Reinforcement Learning for Traffic Signal Control: Demand Surges, Incidents and Sensor Failures","date":"2019-04-17","arxiv_id":"1904.08353","repositories_listed":1,"syntology":null},{"url":"/paper/lets-play-again-variability-of-deep","slug":"lets-play-again-variability-of-deep","title":"Let's Play Again: Variability of Deep Reinforcement Learning Agents in Atari Environments","date":"2019-04-12","arxiv_id":"1904.06312","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-on-a-budget-3d","slug":"deep-reinforcement-learning-on-a-budget-3d","title":"Deep Reinforcement Learning on a Budget: 3D Control and Reasoning Without a Supercomputer","date":"2019-04-03","arxiv_id":"1904.01806","repositories_listed":1,"syntology":null},{"url":"/paper/jointly-pre-training-with-supervised","slug":"jointly-pre-training-with-supervised","title":"Jointly Pre-training with Supervised, Autoencoder, and Value Losses for Deep Reinforcement Learning","date":"2019-04-03","arxiv_id":"1904.02206","repositories_listed":1,"syntology":null},{"url":"/paper/random-projection-in-neural-episodic-control","slug":"random-projection-in-neural-episodic-control","title":"Random Projection in Neural Episodic Control","date":"2019-04-03","arxiv_id":"1904.01790","repositories_listed":1,"syntology":null},{"url":"/paper/autoregressive-policies-for-continuous","slug":"autoregressive-policies-for-continuous","title":"Autoregressive Policies for Continuous Control Deep Reinforcement Learning","date":"2019-03-27","arxiv_id":"1903.11524","repositories_listed":1,"syntology":null},{"url":"/paper/truly-proximal-policy-optimization","slug":"truly-proximal-policy-optimization","title":"Truly Proximal Policy Optimization","date":"2019-03-19","arxiv_id":"1903.07940","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-pitfalls-of-measuring-emergent","slug":"on-the-pitfalls-of-measuring-emergent","title":"On the Pitfalls of Measuring Emergent Communication","date":"2019-03-12","arxiv_id":"1903.05168","repositories_listed":1,"syntology":null},{"url":"/paper/universally-slimmable-networks-and-improved","slug":"universally-slimmable-networks-and-improved","title":"Universally Slimmable Networks and Improved Training Techniques","date":"2019-03-12","arxiv_id":"1903.05134","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/universally-slimmable-networks-and-improved#ran","syntology_url":"https://syntology.ai/paper/1903.05134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.05134"}},"official":{"repos":["JiahuiYu/slimmable_networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-2","slug":"multi-agent-deep-reinforcement-learning-for-2","title":"Multi-Agent Deep Reinforcement Learning for Large-scale Traffic Signal Control","date":"2019-03-11","arxiv_id":"1903.04527","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-deep-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/1903.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04527"}},"official":{"repos":["cts198859/deeprl_signal_control"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-power-system-emergency-control-using","slug":"adaptive-power-system-emergency-control-using","title":"Adaptive Power System Emergency Control using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03712","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-power-system-emergency-control-using#ran","syntology_url":"https://syntology.ai/paper/1903.03712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03712"}},"official":{"repos":["RLGC-Project/RLGC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/viewpoint-optimization-for-autonomous","slug":"viewpoint-optimization-for-autonomous","title":"Viewpoint Optimization for Autonomous Strawberry Harvesting with Deep Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.02074","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-up-budgeted-reinforcement-learning","slug":"scaling-up-budgeted-reinforcement-learning","title":"Budgeted Reinforcement Learning in Continuous State Space","date":"2019-03-03","arxiv_id":"1903.01004","repositories_listed":1,"syntology":null},{"url":"/paper/catalystrl-a-distributed-framework-for","slug":"catalystrl-a-distributed-framework-for","title":"Catalyst.RL: A Distributed Framework for Reproducible RL Research","date":"2019-02-28","arxiv_id":"1903.00027","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-bottlenecks-in-deep-q-learning","slug":"diagnosing-bottlenecks-in-deep-q-learning","title":"Diagnosing Bottlenecks in Deep Q-learning Algorithms","date":"2019-02-26","arxiv_id":"1902.10250","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diagnosing-bottlenecks-in-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1902.10250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.10250"}},"official":null}},{"url":"/paper/flappy-hummingbird-an-open-source-dynamic","slug":"flappy-hummingbird-an-open-source-dynamic","title":"Flappy Hummingbird: An Open Source Dynamic Simulation of Flapping Wing Robots and Animals","date":"2019-02-25","arxiv_id":"1902.09628","repositories_listed":1,"syntology":null},{"url":"/paper/marathon-environments-multi-agent-continuous","slug":"marathon-environments-multi-agent-continuous","title":"Marathon Environments: Multi-Agent Continuous Control Benchmarks in a Modern Video Game Engine","date":"2019-02-25","arxiv_id":"1902.09097","repositories_listed":1,"syntology":null},{"url":"/paper/dom-q-net-grounded-rl-on-structured-language","slug":"dom-q-net-grounded-rl-on-structured-language","title":"DOM-Q-NET: Grounded RL on Structured Language","date":"2019-02-19","arxiv_id":"1902.07257","repositories_listed":1,"syntology":null},{"url":"/paper/prolonets-neural-encoding-human-experts","slug":"prolonets-neural-encoding-human-experts","title":"Neural-encoding Human Experts' Domain Knowledge to Warm Start Reinforcement Learning","date":"2019-02-15","arxiv_id":"1902.06007","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-through-simulation-integrating","slug":"generalization-through-simulation-integrating","title":"Generalization through Simulation: Integrating Simulated and Real Data into Deep Reinforcement Learning for Vision-Based Autonomous Flight","date":"2019-02-11","arxiv_id":"1902.03701","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generalization-through-simulation-integrating#ran","syntology_url":"https://syntology.ai/paper/1902.03701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.03701"}},"official":{"repos":["gkahn13/GtS"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/novelty-search-for-deep-reinforcement","slug":"novelty-search-for-deep-reinforcement","title":"Novelty Search for Deep Reinforcement Learning Policy Network Weights by Action Sequence Edit Metric Distance","date":"2019-02-08","arxiv_id":"1902.03142","repositories_listed":1,"syntology":null},{"url":"/paper/artificial-intelligence-for-prosthetics","slug":"artificial-intelligence-for-prosthetics","title":"Artificial Intelligence for Prosthetics - challenge solutions","date":"2019-02-07","arxiv_id":"1902.02441","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/artificial-intelligence-for-prosthetics#ran","syntology_url":"https://syntology.ai/paper/1902.02441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.02441"}},"official":{"repos":["iasawseen/MultiServerRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-schedule-communication-in-multi","slug":"learning-to-schedule-communication-in-multi","title":"Learning to Schedule Communication in Multi-agent Reinforcement Learning","date":"2019-02-05","arxiv_id":"1902.01554","repositories_listed":1,"syntology":null},{"url":"/paper/policy-consolidation-for-continual","slug":"policy-consolidation-for-continual","title":"Policy Consolidation for Continual Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00255","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-consolidation-for-continual#ran","syntology_url":"https://syntology.ai/paper/1902.00255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00255"}},"official":null}},{"url":"/paper/safe-efficient-and-comfortable-velocity","slug":"safe-efficient-and-comfortable-velocity","title":"Safe, Efficient, and Comfortable Velocity Control based on Reinforcement Learning for Autonomous Driving","date":"2019-01-29","arxiv_id":"1902.00089","repositories_listed":1,"syntology":null},{"url":"/paper/making-deep-q-learning-methods-robust-to-time","slug":"making-deep-q-learning-methods-robust-to-time","title":"Making Deep Q-learning methods robust to time discretization","date":"2019-01-28","arxiv_id":"1901.09732","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-deep-q-learning-methods-robust-to-time#ran","syntology_url":"https://syntology.ai/paper/1901.09732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09732"}},"official":null}},{"url":"/paper/emergent-linguistic-phenomena-in-multi-agent","slug":"emergent-linguistic-phenomena-in-multi-agent","title":"Emergent Linguistic Phenomena in Multi-Agent Communication Games","date":"2019-01-25","arxiv_id":"1901.08706","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/emergent-linguistic-phenomena-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1901.08706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08706"}},"official":{"repos":["lgraesser/MultimodalGame"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/combinational-q-learning-for-dou-di-zhu","slug":"combinational-q-learning-for-dou-di-zhu","title":"Combinational Q-Learning for Dou Di Zhu","date":"2019-01-24","arxiv_id":"1901.08925","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-measurement-scheduling-for-event","slug":"dynamic-measurement-scheduling-for-event","title":"Dynamic Measurement Scheduling for Event Forecasting using Deep RL","date":"2019-01-24","arxiv_id":"1901.09699","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-multi-step-deep-reinforcement","slug":"understanding-multi-step-deep-reinforcement","title":"Understanding Multi-Step Deep Reinforcement Learning: A Systematic Study of the DQN Target","date":"2019-01-22","arxiv_id":"1901.07510","repositories_listed":1,"syntology":null},{"url":"/paper/autophase-compiler-phase-ordering-for-high","slug":"autophase-compiler-phase-ordering-for-high","title":"AutoPhase: Compiler Phase-Ordering for High Level Synthesis with Deep Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04615","repositories_listed":1,"syntology":null},{"url":"/paper/improving-coordination-in-multi-agent-deep","slug":"improving-coordination-in-multi-agent-deep","title":"Improving Coordination in Small-Scale Multi-Agent Deep Reinforcement Learning through Memory-driven Communication","date":"2019-01-12","arxiv_id":"1901.03887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-coordination-in-multi-agent-deep#ran","syntology_url":"https://syntology.ai/paper/1901.03887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.03887"}},"official":null}},{"url":"/paper/self-learning-exploration-and-mapping-for","slug":"self-learning-exploration-and-mapping-for","title":"Self-Learning Exploration and Mapping for Mobile Robots via Deep Reinforcement Learning","date":"2019-01-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-interpret-atari-agents","slug":"learn-to-interpret-atari-agents","title":"Learn to Interpret Atari Agents","date":"2018-12-29","arxiv_id":"1812.11276","repositories_listed":1,"syntology":null},{"url":"/paper/iroko-a-framework-to-prototype-reinforcement","slug":"iroko-a-framework-to-prototype-reinforcement","title":"Iroko: A Framework to Prototype Reinforcement Learning for Data Center Traffic Control","date":"2018-12-24","arxiv_id":"1812.09975","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-with-non-expert-human","slug":"pre-training-with-non-expert-human","title":"Pre-training with Non-expert Human Demonstration for Deep Reinforcement Learning","date":"2018-12-21","arxiv_id":"1812.08904","repositories_listed":1,"syntology":null},{"url":"/paper/information-directed-exploration-for-deep","slug":"information-directed-exploration-for-deep","title":"Information-Directed Exploration for Deep Reinforcement Learning","date":"2018-12-18","arxiv_id":"1812.07544","repositories_listed":1,"syntology":null},{"url":"/paper/an-atari-model-zoo-for-analyzing-visualizing","slug":"an-atari-model-zoo-for-analyzing-visualizing","title":"An Atari Model Zoo for Analyzing, Visualizing, and Comparing Deep Reinforcement Learning Agents","date":"2018-12-17","arxiv_id":"1812.07069","repositories_listed":1,"syntology":null},{"url":"/paper/residual-policy-learning","slug":"residual-policy-learning","title":"Residual Policy Learning","date":"2018-12-15","arxiv_id":"1812.06298","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/residual-policy-learning#ran","syntology_url":"https://syntology.ai/paper/1812.06298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.06298"}},"official":null}},{"url":"/paper/simulation-to-scaled-city-zero-shot-policy","slug":"simulation-to-scaled-city-zero-shot-policy","title":"Simulation to Scaled City: Zero-Shot Policy Transfer for Traffic Control via Autonomous Vehicles","date":"2018-12-14","arxiv_id":"1812.06120","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-conscious-reinforcement-learning","slug":"exploration-conscious-reinforcement-learning","title":"Exploration Conscious Reinforcement Learning Revisited","date":"2018-12-13","arxiv_id":"1812.05551","repositories_listed":1,"syntology":null},{"url":"/paper/pseudo-rehearsal-achieving-deep-reinforcement","slug":"pseudo-rehearsal-achieving-deep-reinforcement","title":"Pseudo-Rehearsal: Achieving Deep Reinforcement Learning without Catastrophic Forgetting","date":"2018-12-06","arxiv_id":"1812.02464","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-generalization-in-reinforcement","slug":"quantifying-generalization-in-reinforcement","title":"Quantifying Generalization in Reinforcement Learning","date":"2018-12-06","arxiv_id":"1812.02341","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-generalization-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1812.02341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.02341"}},"official":{"repos":["openai/coinrun"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/playing-text-adventure-games-with-graph-based","slug":"playing-text-adventure-games-with-graph-based","title":"Playing Text-Adventure Games with Graph-Based Deep Reinforcement Learning","date":"2018-12-04","arxiv_id":"1812.01628","repositories_listed":1,"syntology":null},{"url":"/paper/towards-solving-text-based-games-by-producing","slug":"towards-solving-text-based-games-by-producing","title":"Towards Solving Text-based Games by Producing Adaptive Action Spaces","date":"2018-12-03","arxiv_id":"1812.00855","repositories_listed":1,"syntology":null},{"url":"/paper/visual-foresight-model-based-deep","slug":"visual-foresight-model-based-deep","title":"Visual Foresight: Model-Based Deep Reinforcement Learning for Vision-Based Robotic Control","date":"2018-12-03","arxiv_id":"1812.00568","repositories_listed":1,"syntology":null},{"url":"/paper/macro-action-selection-with-deep","slug":"macro-action-selection-with-deep","title":"Macro action selection with deep reinforcement learning in StarCraft","date":"2018-12-02","arxiv_id":"1812.00336","repositories_listed":1,"syntology":null},{"url":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/deep-multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1811.12557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.12557"}},"official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-autonomous","slug":"deep-reinforcement-learning-for-autonomous","title":"Deep Reinforcement Learning for Autonomous Driving","date":"2018-11-28","arxiv_id":"1811.11329","repositories_listed":1,"syntology":null},{"url":"/paper/hardware-conditioned-policies-for-multi-robot","slug":"hardware-conditioned-policies-for-multi-robot","title":"Hardware Conditioned Policies for Multi-Robot Transfer Learning","date":"2018-11-24","arxiv_id":"1811.09864","repositories_listed":1,"syntology":null},{"url":"/paper/urban-driving-with-multi-objective-deep","slug":"urban-driving-with-multi-objective-deep","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","date":"2018-11-21","arxiv_id":"1811.08586","repositories_listed":1,"syntology":null},{"url":"/paper/trolleymod-v10-an-open-source-simulation-and","slug":"trolleymod-v10-an-open-source-simulation-and","title":"TrolleyMod v1.0: An Open-Source Simulation and Data-Collection Platform for Ethical Decision Making in Autonomous Vehicles","date":"2018-11-14","arxiv_id":"1811.05594","repositories_listed":1,"syntology":null},{"url":"/paper/saferoute-learning-to-navigate-streets-safely","slug":"saferoute-learning-to-navigate-streets-safely","title":"SafeRoute: Learning to Navigate Streets Safely in an Urban Environment","date":"2018-11-03","arxiv_id":"1811.01147","repositories_listed":1,"syntology":null},{"url":"/paper/3d-traffic-simulation-for-autonomous-vehicles","slug":"3d-traffic-simulation-for-autonomous-vehicles","title":"3D Traffic Simulation for Autonomous Vehicles in Unity and Python","date":"2018-10-30","arxiv_id":"1810.12552","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-and-deep-learning","slug":"reinforcement-learning-and-deep-learning","title":"Reinforcement Learning and Deep Learning based Lateral Control for Autonomous Driving","date":"2018-10-30","arxiv_id":"1810.12778","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-generalization-in-deep","slug":"assessing-generalization-in-deep","title":"Assessing Generalization in Deep Reinforcement Learning","date":"2018-10-29","arxiv_id":"1810.12282","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assessing-generalization-in-deep#ran","syntology_url":"https://syntology.ai/paper/1810.12282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12282"}},"official":{"repos":["sunblaze-ucb/rl-generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-inference-with-tail-adaptive-f","slug":"variational-inference-with-tail-adaptive-f","title":"Variational Inference with Tail-adaptive f-Divergence","date":"2018-10-29","arxiv_id":"1810.11943","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/variational-inference-with-tail-adaptive-f#ran","syntology_url":"https://syntology.ai/paper/1810.11943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11943"}},"official":{"repos":["dilinwang820/adaptive-f-divergence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/learn-to-steer-through-deep-reinforcement","slug":"learn-to-steer-through-deep-reinforcement","title":"Learn to Steer through Deep Reinforcement Learning","date":"2018-10-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/transfer-of-deep-reactive-policies-for-mdp","slug":"transfer-of-deep-reactive-policies-for-mdp","title":"Transfer of Deep Reactive Policies for MDP Planning","date":"2018-10-26","arxiv_id":"1810.11488","repositories_listed":1,"syntology":null},{"url":"/paper/inverse-reinforcement-learning-for-video","slug":"inverse-reinforcement-learning-for-video","title":"Inverse reinforcement learning for video games","date":"2018-10-24","arxiv_id":"1810.10593","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-eligibility-traces-for-deep","slug":"efficient-eligibility-traces-for-deep","title":"Reconciling $λ$-Returns with Experience Replay","date":"2018-10-23","arxiv_id":"1810.09967","repositories_listed":1,"syntology":null},{"url":"/paper/rlgraph-modular-computation-graphs-for-deep","slug":"rlgraph-modular-computation-graphs-for-deep","title":"RLgraph: Modular Computation Graphs for Deep Reinforcement Learning","date":"2018-10-21","arxiv_id":"1810.09028","repositories_listed":1,"syntology":null},{"url":"/paper/social-behavior-learning-with-realistic","slug":"social-behavior-learning-with-realistic","title":"Learning Socially Appropriate Robot Approaching Behavior Toward Groups using Deep Reinforcement Learning","date":"2018-10-16","arxiv_id":"1810.06979","repositories_listed":1,"syntology":null},{"url":"/paper/visual-semantic-navigation-using-scene-priors","slug":"visual-semantic-navigation-using-scene-priors","title":"Visual Semantic Navigation using Scene Priors","date":"2018-10-15","arxiv_id":"1810.06543","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-potential-of-classical-q","slug":"assessing-the-potential-of-classical-q","title":"Assessing the Potential of Classical Q-learning in General Game Playing","date":"2018-10-14","arxiv_id":"1810.06078","repositories_listed":1,"syntology":null},{"url":"/paper/empowerment-driven-exploration-using-mutual","slug":"empowerment-driven-exploration-using-mutual","title":"Empowerment-driven Exploration using Mutual Information Estimation","date":"2018-10-11","arxiv_id":"1810.05533","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-deep-reinforcement-learning","slug":"semi-supervised-deep-reinforcement-learning","title":"Semi-supervised Deep Reinforcement Learning in Support of IoT and Smart City Services","date":"2018-10-09","arxiv_id":"1810.04118","repositories_listed":1,"syntology":null}],"record_sha256":"ccbe40df712aaf7dfb7199ca95e72e52e3a44cfcccea35c836fea7435d805163","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}