{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/38","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":38,"pages_in_order":152,"rows_per_page":100,"rows":[3701,3800],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/37","next":"/task/reinforcement-learning-1/papers/39","papers":[{"url":"/paper/an-asymptotically-optimal-multi-armed-bandit","slug":"an-asymptotically-optimal-multi-armed-bandit","title":"An Asymptotically Optimal Multi-Armed Bandit Algorithm and Hyperparameter Optimization","date":"2020-07-11","arxiv_id":"2007.05670","repositories_listed":1,"syntology":null},{"url":"/paper/long-term-planning-with-deep-reinforcement","slug":"long-term-planning-with-deep-reinforcement","title":"Long-Term Planning with Deep Reinforcement Learning on Autonomous Drones","date":"2020-07-11","arxiv_id":"2007.05694","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/long-term-planning-with-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.05694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05694"}},"official":{"repos":["ugurkanates/NeurIRS2019DroneChallengeRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-trained-word-embeddings-for-goal","slug":"pre-trained-word-embeddings-for-goal","title":"Pre-trained Word Embeddings for Goal-conditional Transfer Learning in Reinforcement Learning","date":"2020-07-10","arxiv_id":"2007.05196","repositories_listed":1,"syntology":null},{"url":"/paper/fast-reinforcement-learning-with-generalized","slug":"fast-reinforcement-learning-with-generalized","title":"Fast reinforcement learning with generalized policy updates","date":"2020-07-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-retrospective-knowledge-with-reverse","slug":"learning-retrospective-knowledge-with-reverse","title":"Learning Retrospective Knowledge with Reverse Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.06703","repositories_listed":1,"syntology":null},{"url":"/paper/sunrise-a-simple-unified-framework-for","slug":"sunrise-a-simple-unified-framework-for","title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04938","repositories_listed":1,"syntology":null},{"url":"/paper/provably-safe-pac-mdp-exploration-using","slug":"provably-safe-pac-mdp-exploration-using","title":"Provably Safe PAC-MDP Exploration Using Analogies","date":"2020-07-07","arxiv_id":"2007.03574","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-data-augmentation-using","slug":"counterfactual-data-augmentation-using","title":"Counterfactual Data Augmentation using Locally Factored Dynamics","date":"2020-07-06","arxiv_id":"2007.02863","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-data-augmentation-using#ran","syntology_url":"https://syntology.ai/paper/2007.02863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02863"}},"official":null}},{"url":"/paper/enhancing-sat-solvers-with-glue-variable","slug":"enhancing-sat-solvers-with-glue-variable","title":"Enhancing SAT solvers with glue variable predictions","date":"2020-07-06","arxiv_id":"2007.02559","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":2,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-sat-solvers-with-glue-variable#ran","syntology_url":"https://syntology.ai/paper/2007.02559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02559"}},"official":{"repos":["jesse-michael-han/neuro-cadical"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-implicit-credit-assignment-for-multi","slug":"learning-implicit-credit-assignment-for-multi","title":"Learning Implicit Credit Assignment for Cooperative Multi-Agent Reinforcement Learning","date":"2020-07-06","arxiv_id":"2007.02529","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-implicit-credit-assignment-for-multi#ran","syntology_url":"https://syntology.ai/paper/2007.02529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02529"}},"official":{"repos":["mzho7212/LICA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/lfq-online-learning-of-per-flow-queuing","slug":"lfq-online-learning-of-per-flow-queuing","title":"LFQ: Online Learning of Per-flow Queuing Policies using Deep Reinforcement Learning","date":"2020-07-06","arxiv_id":"2007.02735","repositories_listed":1,"syntology":null},{"url":"/paper/nappo-modular-and-scalable-reinforcement","slug":"nappo-modular-and-scalable-reinforcement","title":"Integrating Distributed Architectures in Highly Modular RL Libraries","date":"2020-07-06","arxiv_id":"2007.02622","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/nappo-modular-and-scalable-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.02622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02622"}},"official":null}},{"url":"/paper/bidirectional-model-based-policy-optimization","slug":"bidirectional-model-based-policy-optimization","title":"Bidirectional Model-based Policy Optimization","date":"2020-07-04","arxiv_id":"2007.01995","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bidirectional-model-based-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2007.01995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01995"}},"official":{"repos":["hanglai/bmpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discount-factor-as-a-regularizer-in","slug":"discount-factor-as-a-regularizer-in","title":"Discount Factor as a Regularizer in Reinforcement Learning","date":"2020-07-04","arxiv_id":"2007.02040","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discount-factor-as-a-regularizer-in#ran","syntology_url":"https://syntology.ai/paper/2007.02040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02040"}},"official":{"repos":["ron-amit/Discount_as_Regularizer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/e-bmc-a-bayesian-ensemble-approach-to-epsilon","slug":"e-bmc-a-bayesian-ensemble-approach-to-epsilon","title":"ε-BMC: A Bayesian Ensemble Approach to Epsilon-Greedy Exploration in Model-Free Reinforcement Learning","date":"2020-07-02","arxiv_id":"2007.00869","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-search-efficiently-for-causally","slug":"learning-to-search-efficiently-for-causally","title":"Learning to search efficiently for causally near-optimal treatments","date":"2020-07-02","arxiv_id":"2007.00973","repositories_listed":1,"syntology":null},{"url":"/paper/robust-inverse-reinforcement-learning-under","slug":"robust-inverse-reinforcement-learning-under","title":"Robust Inverse Reinforcement Learning under Transition Dynamics Mismatch","date":"2020-07-02","arxiv_id":"2007.01174","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/robust-inverse-reinforcement-learning-under#ran","syntology_url":"https://syntology.ai/paper/2007.01174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01174"}},"official":{"repos":["lviano/robustmce_irl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/verifiably-safe-exploration-for-end-to-end","slug":"verifiably-safe-exploration-for-end-to-end","title":"Verifiably Safe Exploration for End-to-End Reinforcement Learning","date":"2020-07-02","arxiv_id":"2007.01223","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-discretization-for-model-based","slug":"adaptive-discretization-for-model-based","title":"Adaptive Discretization for Model-Based Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.00717","repositories_listed":1,"syntology":null},{"url":"/paper/debiased-contrastive-learning","slug":"debiased-contrastive-learning","title":"Debiased Contrastive Learning","date":"2020-07-01","arxiv_id":"2007.00224","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/debiased-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2007.00224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.00224"}},"official":{"repos":["chingyaoc/DCL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/group-equivariant-deep-reinforcement-learning","slug":"group-equivariant-deep-reinforcement-learning","title":"Group Equivariant Deep Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.03437","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/group-equivariant-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.03437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03437"}},"official":{"repos":["arnab39/EquivariantDQN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reinforcement-learning-based-control-of","slug":"reinforcement-learning-based-control-of","title":"Reinforcement Learning based Control of Imitative Policies for Near-Accident Driving","date":"2020-07-01","arxiv_id":"2007.00178","repositories_listed":1,"syntology":null},{"url":"/paper/deep-feature-space-a-geometrical-perspective","slug":"deep-feature-space-a-geometrical-perspective","title":"Deep Feature Space: A Geometrical Perspective","date":"2020-06-30","arxiv_id":"2007.00062","repositories_listed":1,"syntology":null},{"url":"/paper/enforcing-almost-sure-reachability-in-pomdps","slug":"enforcing-almost-sure-reachability-in-pomdps","title":"Enforcing Almost-Sure Reachability in POMDPs","date":"2020-06-30","arxiv_id":"2007.00085","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-performance-of-reinforcement","slug":"evaluating-the-performance-of-reinforcement","title":"Evaluating the Performance of Reinforcement Learning Algorithms","date":"2020-06-30","arxiv_id":"2006.16958","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-for-semi","slug":"model-based-reinforcement-learning-for-semi","title":"Model-based Reinforcement Learning for Semi-Markov Decision Processes with Neural ODEs","date":"2020-06-29","arxiv_id":"2006.16210","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-semi#ran","syntology_url":"https://syntology.ai/paper/2006.16210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16210"}},"official":{"repos":["dtak/mbrl-smdp-ode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/image-classification-by-reinforcement","slug":"image-classification-by-reinforcement","title":"Image Classification by Reinforcement Learning with Two-State Q-Learning","date":"2020-06-28","arxiv_id":"2007.01298","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforced-model-for-zero-shot-cross-1","slug":"a-deep-reinforced-model-for-zero-shot-cross-1","title":"A Deep Reinforced Model for Zero-Shot Cross-Lingual Summarization with Bilingual Semantic Similarity Rewards","date":"2020-06-27","arxiv_id":"2006.15454","repositories_listed":1,"syntology":null},{"url":"/paper/explanation-augmented-feedback-in-human-in","slug":"explanation-augmented-feedback-in-human-in","title":"Widening the Pipeline in Human-Guided Reinforcement Learning with Explanation and Context-Aware Data Augmentation","date":"2020-06-26","arxiv_id":"2006.14804","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-reward-driven-imitation-learning","slug":"intrinsic-reward-driven-imitation-learning","title":"Intrinsic Reward Driven Imitation Learning via Generative Model","date":"2020-06-26","arxiv_id":"2006.15061","repositories_listed":1,"syntology":null},{"url":"/paper/online-3d-bin-packing-with-constrained-deep","slug":"online-3d-bin-packing-with-constrained-deep","title":"Online 3D Bin Packing with Constrained Deep Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.14978","repositories_listed":1,"syntology":null},{"url":"/paper/policy-gnn-aggregation-optimization-for-graph","slug":"policy-gnn-aggregation-optimization-for-graph","title":"Policy-GNN: Aggregation Optimization for Graph Neural Networks","date":"2020-06-26","arxiv_id":"2006.15097","repositories_listed":1,"syntology":null},{"url":"/paper/what-can-i-do-here-a-theory-of-affordances-in","slug":"what-can-i-do-here-a-theory-of-affordances-in","title":"What can I do here? A Theory of Affordances in Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.15085","repositories_listed":1,"syntology":null},{"url":"/paper/newton-type-methods-for-minimax-optimization","slug":"newton-type-methods-for-minimax-optimization","title":"Newton-type Methods for Minimax Optimization","date":"2020-06-25","arxiv_id":"2006.14592","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-data-augmentation-for","slug":"automatic-data-augmentation-for","title":"Automatic Data Augmentation for Generalization in Deep Reinforcement Learning","date":"2020-06-23","arxiv_id":"2006.12862","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automatic-data-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2006.12862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.12862"}},"official":{"repos":["rraileanu/auto-drac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/experience-replay-with-likelihood-free","slug":"experience-replay-with-likelihood-free","title":"Experience Replay with Likelihood-free Importance Weights","date":"2020-06-23","arxiv_id":"2006.13169","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/experience-replay-with-likelihood-free#ran","syntology_url":"https://syntology.ai/paper/2006.13169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13169"}},"official":null}},{"url":"/paper/dm-control-software-and-tasks-for-continuous","slug":"dm-control-software-and-tasks-for-continuous","title":"dm_control: Software and Tasks for Continuous Control","date":"2020-06-22","arxiv_id":"2006.12983","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-via-curriculum","slug":"safe-reinforcement-learning-via-curriculum","title":"Safe Reinforcement Learning via Curriculum Induction","date":"2020-06-22","arxiv_id":"2006.12136","repositories_listed":1,"syntology":null},{"url":"/paper/automated-optical-multi-layer-design-via-deep","slug":"automated-optical-multi-layer-design-via-deep","title":"Automated Optical Multi-layer Design via Deep Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11940","repositories_listed":1,"syntology":null},{"url":"/paper/generating-adjacency-constrained-subgoals-in","slug":"generating-adjacency-constrained-subgoals-in","title":"Generating Adjacency-Constrained Subgoals in Hierarchical Reinforcement Learning","date":"2020-06-20","arxiv_id":"2006.11485","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generating-adjacency-constrained-subgoals-in#ran","syntology_url":"https://syntology.ai/paper/2006.11485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11485"}},"official":{"repos":["trzhang0116/HRAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/deep-implicit-coordination-graphs-for-multi","slug":"deep-implicit-coordination-graphs-for-multi","title":"Deep Implicit Coordination Graphs for Multi-agent Reinforcement Learning","date":"2020-06-19","arxiv_id":"2006.11438","repositories_listed":1,"syntology":null},{"url":"/paper/task-agnostic-online-reinforcement-learning","slug":"task-agnostic-online-reinforcement-learning","title":"Task-Agnostic Online Reinforcement Learning with an Infinite Mixture of Gaussian Processes","date":"2020-06-19","arxiv_id":"2006.11441","repositories_listed":1,"syntology":null},{"url":"/paper/dream-deep-regret-minimization-with-advantage","slug":"dream-deep-regret-minimization-with-advantage","title":"DREAM: Deep Regret minimization with Advantage baselines and Model-free learning","date":"2020-06-18","arxiv_id":"2006.10410","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-ridesharing-dispatch-using-multi","slug":"efficient-ridesharing-dispatch-using-multi","title":"Efficient Ridesharing Dispatch Using Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10897","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-curriculum-learning-through-value","slug":"automatic-curriculum-learning-through-value","title":"Automatic Curriculum Learning through Value Disagreement","date":"2020-06-17","arxiv_id":"2006.09641","repositories_listed":1,"syntology":null},{"url":"/paper/delta-schema-network-in-model-based","slug":"delta-schema-network-in-model-based","title":"Delta Schema Network in Model-based Reinforcement Learning","date":"2020-06-17","arxiv_id":"2006.09950","repositories_listed":1,"syntology":null},{"url":"/paper/forgetful-experience-replay-in-hierarchical","slug":"forgetful-experience-replay-in-hierarchical","title":"Forgetful Experience Replay in Hierarchical Reinforcement Learning from Demonstrations","date":"2020-06-17","arxiv_id":"2006.09939","repositories_listed":1,"syntology":null},{"url":"/paper/green-simulation-assisted-reinforcement","slug":"green-simulation-assisted-reinforcement","title":"Green Simulation Assisted Reinforcement Learning with Model Risk for Biomanufacturing Learning and Control","date":"2020-06-17","arxiv_id":"2006.09919","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-track-dynamic-targets-in","slug":"learning-to-track-dynamic-targets-in","title":"Learning to Track Dynamic Targets in Partially Known Environments","date":"2020-06-17","arxiv_id":"2006.10190","repositories_listed":1,"syntology":null},{"url":"/paper/nnc-neural-network-control-of-dynamical","slug":"nnc-neural-network-control-of-dynamical","title":"Neural Ordinary Differential Equation Control of Dynamics on Graphs","date":"2020-06-17","arxiv_id":"2006.09773","repositories_listed":1,"syntology":{"n":18,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nnc-neural-network-control-of-dynamical#ran","syntology_url":"https://syntology.ai/paper/2006.09773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09773"}},"official":{"repos":["asikist/nnc"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-adversarial-meta-reinforcement","slug":"model-based-adversarial-meta-reinforcement","title":"Model-based Adversarial Meta-Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.08875","repositories_listed":1,"syntology":null},{"url":"/paper/opponent-modelling-with-local-information","slug":"opponent-modelling-with-local-information","title":"Agent Modelling under Partial Observability for Deep Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.09447","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/opponent-modelling-with-local-information#ran","syntology_url":"https://syntology.ai/paper/2006.09447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09447"}},"official":{"repos":["uoe-agents/LIAM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-based-value-functions","slug":"parameter-based-value-functions","title":"Parameter-Based Value Functions","date":"2020-06-16","arxiv_id":"2006.09226","repositories_listed":1,"syntology":null},{"url":"/paper/robot-perception-enables-complex-navigation","slug":"robot-perception-enables-complex-navigation","title":"Robot Perception enables Complex Navigation Behavior via Self-Supervised Learning","date":"2020-06-16","arxiv_id":"2006.08967","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-model-based-reinforcement-learning-1","slug":"efficient-model-based-reinforcement-learning-1","title":"Efficient Model-Based Reinforcement Learning through Optimistic Policy Search and Planning","date":"2020-06-15","arxiv_id":"2006.08684","repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-effectively-explore-in-context-based","slug":"learn-to-effectively-explore-in-context-based","title":"MetaCURE: Meta Reinforcement Learning with Empowerment-Driven Exploration","date":"2020-06-15","arxiv_id":"2006.08170","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learn-to-effectively-explore-in-context-based#ran","syntology_url":"https://syntology.ai/paper/2006.08170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08170"}},"official":{"repos":["NagisaZj/MetaCURE-Public"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiagent-reinforcement-learning-based","slug":"multiagent-reinforcement-learning-based","title":"Multiagent Reinforcement Learning based Energy Beamforming Control","date":"2020-06-15","arxiv_id":"2006.08829","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-distributionally-robust-policy","slug":"optimistic-distributionally-robust-policy","title":"Optimistic Distributionally Robust Policy Optimization","date":"2020-06-14","arxiv_id":"2006.07815","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimistic-distributionally-robust-policy#ran","syntology_url":"https://syntology.ai/paper/2006.07815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07815"}},"official":{"repos":["kadysongbb/dr-trpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exchangeable-models-in-meta-reinforcement","slug":"exchangeable-models-in-meta-reinforcement","title":"Exchangeable Models in Meta Reinforcement Learning","date":"2020-06-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mutual-information-based-knowledge-transfer","slug":"mutual-information-based-knowledge-transfer","title":"Mutual Information Based Knowledge Transfer Under State-Action Dimension Mismatch","date":"2020-06-12","arxiv_id":"2006.07041","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-sum-product-max-networks-for","slug":"recurrent-sum-product-max-networks-for","title":"Recurrent Sum-Product-Max Networks for Decision Making in Perfectly-Observed Environments","date":"2020-06-12","arxiv_id":"2006.07300","repositories_listed":1,"syntology":null},{"url":"/paper/samba-safe-model-based-active-reinforcement","slug":"samba-safe-model-based-active-reinforcement","title":"SAMBA: Safe Model-Based & Active Reinforcement Learning","date":"2020-06-12","arxiv_id":"2006.09436","repositories_listed":1,"syntology":null},{"url":"/paper/closed-loop-neural-symbolic-learning-via","slug":"closed-loop-neural-symbolic-learning-via","title":"Closed Loop Neural-Symbolic Learning via Integrating Neural Perception, Grammar Parsing, and Symbolic Reasoning","date":"2020-06-11","arxiv_id":"2006.06649","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/closed-loop-neural-symbolic-learning-via#ran","syntology_url":"https://syntology.ai/paper/2006.06649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06649"}},"official":{"repos":["liqing-ustc/NGS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distributed-reinforcement-learning-in-multi","slug":"distributed-reinforcement-learning-in-multi","title":"Multi-Agent Reinforcement Learning in Stochastic Networked Systems","date":"2020-06-11","arxiv_id":"2006.06555","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distributed-reinforcement-learning-in-multi#ran","syntology_url":"https://syntology.ai/paper/2006.06555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06555"}},"official":null}},{"url":"/paper/reinforcement-learning-from-a-mixture-of","slug":"reinforcement-learning-from-a-mixture-of","title":"Continuous Action Reinforcement Learning from a Mixture of Interpretable Experts","date":"2020-06-10","arxiv_id":"2006.05911","repositories_listed":1,"syntology":null},{"url":"/paper/robust-detection-of-adaptive-spammers-by-nash","slug":"robust-detection-of-adaptive-spammers-by-nash","title":"Robust Spammer Detection by Nash Reinforcement Learning","date":"2020-06-10","arxiv_id":"2006.06069","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/robust-detection-of-adaptive-spammers-by-nash#ran","syntology_url":"https://syntology.ai/paper/2006.06069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06069"}},"official":{"repos":["YingtongDou/Nash-Detect"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/constrained-episodic-reinforcement-learning","slug":"constrained-episodic-reinforcement-learning","title":"Constrained episodic reinforcement learning in concave-convex and knapsack settings","date":"2020-06-09","arxiv_id":"2006.05051","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/constrained-episodic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.05051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05051"}},"official":{"repos":["miryoosefi/ConRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/online-learning-in-iterated-prisoner-s","slug":"online-learning-in-iterated-prisoner-s","title":"Online Learning in Iterated Prisoner's Dilemma to Mimic Human Behavior","date":"2020-06-09","arxiv_id":"2006.06580","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-under-moral","slug":"reinforcement-learning-under-moral","title":"Reinforcement Learning Under Moral Uncertainty","date":"2020-06-08","arxiv_id":"2006.04734","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-under-moral#ran","syntology_url":"https://syntology.ai/paper/2006.04734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04734"}},"official":{"repos":["uber-research/normative-uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tools-for-data-driven-modeling-of-within-hand","slug":"tools-for-data-driven-modeling-of-within-hand","title":"Tools for Data-driven Modeling of Within-Hand Manipulation with Underactuated Adaptive Hands","date":"2020-06-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dual-policy-distillation","slug":"dual-policy-distillation","title":"Dual Policy Distillation","date":"2020-06-07","arxiv_id":"2006.04061","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/dual-policy-distillation#ran","syntology_url":"https://syntology.ai/paper/2006.04061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04061"}},"official":null}},{"url":"/paper/reinforcement-learning-for-multi-product","slug":"reinforcement-learning-for-multi-product","title":"Reinforcement Learning for Multi-Product Multi-Node Inventory Management in Supply Chains","date":"2020-06-07","arxiv_id":"2006.04037","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-for-multi-product#ran","syntology_url":"https://syntology.ai/paper/2006.04037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04037"}},"official":null}},{"url":"/paper/proximal-gradient-temporal-difference","slug":"proximal-gradient-temporal-difference","title":"Proximal Gradient Temporal Difference Learning: Stable Reinforcement Learning with Polynomial Sample Complexity","date":"2020-06-06","arxiv_id":"2006.03976","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-killed-the-cat-and-the","slug":"curiosity-killed-the-cat-and-the","title":"Curiosity Killed or Incapacitated the Cat and the Asymptotically Optimal Agent","date":"2020-06-05","arxiv_id":"2006.03357","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-update-mechanism-for-q-networks-based","slug":"a-novel-update-mechanism-for-q-networks-based","title":"A Novel Update Mechanism for Q-Networks Based On Extreme Learning Machines","date":"2020-06-04","arxiv_id":"2006.02986","repositories_listed":1,"syntology":null},{"url":"/paper/optimization-and-passive-flow-control-using","slug":"optimization-and-passive-flow-control-using","title":"Single-step deep reinforcement learning for open-loop control of laminar and turbulent flows","date":"2020-06-04","arxiv_id":"2006.02979","repositories_listed":1,"syntology":null},{"url":"/paper/solving-hard-ai-planning-instances-using","slug":"solving-hard-ai-planning-instances-using","title":"Solving Hard AI Planning Instances Using Curriculum-Driven Deep Reinforcement Learning","date":"2020-06-04","arxiv_id":"2006.02689","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/solving-hard-ai-planning-instances-using#ran","syntology_url":"https://syntology.ai/paper/2006.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.02689"}},"official":null}},{"url":"/paper/visual-transfer-for-reinforcement-learning","slug":"visual-transfer-for-reinforcement-learning","title":"Visual Transfer for Reinforcement Learning via Wasserstein Domain Confusion","date":"2020-06-04","arxiv_id":"2006.03465","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-transfer-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03465"}},"official":null}},{"url":"/paper/interferobot-aligning-an-optical","slug":"interferobot-aligning-an-optical","title":"Interferobot: aligning an optical interferometer by a reinforcement learning agent","date":"2020-06-03","arxiv_id":"2006.02252","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-and-1","slug":"combining-reinforcement-learning-and-1","title":"Combining Reinforcement Learning and Constraint Programming for Combinatorial Optimization","date":"2020-06-02","arxiv_id":"2006.01610","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-actor-critic-sample-aware-entropy","slug":"diversity-actor-critic-sample-aware-entropy","title":"Diversity Actor-Critic: Sample-Aware Entropy Regularization for Sample-Efficient Exploration","date":"2020-06-02","arxiv_id":"2006.01419","repositories_listed":1,"syntology":null},{"url":"/paper/learning-optimal-environments-using-projected","slug":"learning-optimal-environments-using-projected","title":"Jointly Learning Environments and Control Policies with Projected Stochastic Gradient Ascent","date":"2020-06-02","arxiv_id":"2006.01738","repositories_listed":1,"syntology":null},{"url":"/paper/temporally-extended-e-greedy-exploration","slug":"temporally-extended-e-greedy-exploration","title":"Temporally-Extended ε-Greedy Exploration","date":"2020-06-02","arxiv_id":"2006.01782","repositories_listed":1,"syntology":null},{"url":"/paper/encoding-formulas-as-deep-networks","slug":"encoding-formulas-as-deep-networks","title":"Encoding formulas as deep networks: Reinforcement learning for zero-shot execution of LTL formulas","date":"2020-06-01","arxiv_id":"2006.01110","repositories_listed":1,"syntology":null},{"url":"/paper/invariant-policy-optimization-towards","slug":"invariant-policy-optimization-towards","title":"Invariant Policy Optimization: Towards Stronger Generalization in Reinforcement Learning","date":"2020-06-01","arxiv_id":"2006.01096","repositories_listed":1,"syntology":null},{"url":"/paper/plangan-model-based-planning-with-sparse","slug":"plangan-model-based-planning-with-sparse","title":"PlanGAN: Model-based Planning With Sparse Rewards and Multiple Goals","date":"2020-06-01","arxiv_id":"2006.00900","repositories_listed":1,"syntology":null},{"url":"/paper/mm-ktd-multiple-model-kalman-temporal","slug":"mm-ktd-multiple-model-kalman-temporal","title":"MM-KTD: Multiple Model Kalman Temporal Differences for Reinforcement Learning","date":"2020-05-30","arxiv_id":"2006.00195","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning","slug":"reinforcement-learning","title":"Reinforcement Learning","date":"2020-05-29","arxiv_id":"2005.14419","repositories_listed":1,"syntology":null},{"url":"/paper/sim2real-for-peg-hole-insertion-with-eye-in","slug":"sim2real-for-peg-hole-insertion-with-eye-in","title":"Sim2Real for Peg-Hole Insertion with Eye-in-Hand Camera","date":"2020-05-29","arxiv_id":"2005.14401","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-sharing-is-surprisingly-useful-for","slug":"parameter-sharing-is-surprisingly-useful-for","title":"Revisiting Parameter Sharing in Multi-Agent Deep Reinforcement Learning","date":"2020-05-27","arxiv_id":"2005.13625","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to-rare","slug":"a-reinforcement-learning-approach-to-rare","title":"A reinforcement learning approach to rare trajectory sampling","date":"2020-05-26","arxiv_id":"2005.12890","repositories_listed":1,"syntology":null},{"url":"/paper/alba-reinforcement-learning-for-video-object","slug":"alba-reinforcement-learning-for-video-object","title":"ALBA : Reinforcement Learning for Video Object Segmentation","date":"2020-05-26","arxiv_id":"2005.13039","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-penetration-testing-with","slug":"modeling-penetration-testing-with","title":"Modeling Penetration Testing with Reinforcement Learning Using Capture-the-Flag Challenges: Trade-offs between Model-free Learning and A Priori Knowledge","date":"2020-05-26","arxiv_id":"2005.12632","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-discovery-of-interpretable-planning","slug":"automatic-discovery-of-interpretable-planning","title":"Automatic Discovery of Interpretable Planning Strategies","date":"2020-05-24","arxiv_id":"2005.11730","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-deep-reinforcement-learning-for","slug":"decentralized-deep-reinforcement-learning-for","title":"Decentralized Deep Reinforcement Learning for a Distributed and Adaptive Locomotion Controller of a Hexapod Robot","date":"2020-05-21","arxiv_id":"2005.11164","repositories_listed":1,"syntology":null},{"url":"/paper/novel-policy-seeking-with-constrained","slug":"novel-policy-seeking-with-constrained","title":"Novel Policy Seeking with Constrained Optimization","date":"2020-05-21","arxiv_id":"2005.10696","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-descent-policy-optimization","slug":"mirror-descent-policy-optimization","title":"Mirror Descent Policy Optimization","date":"2020-05-20","arxiv_id":"2005.09814","repositories_listed":1,"syntology":null},{"url":"/paper/ultrasound-video-summarization-using-deep","slug":"ultrasound-video-summarization-using-deep","title":"Ultrasound Video Summarization using Deep Reinforcement Learning","date":"2020-05-19","arxiv_id":"2005.09531","repositories_listed":1,"syntology":null},{"url":"/paper/local-and-global-explanations-of-agent","slug":"local-and-global-explanations-of-agent","title":"Local and Global Explanations of Agent Behavior: Integrating Strategy Summaries with Saliency Maps","date":"2020-05-18","arxiv_id":"2005.08874","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/local-and-global-explanations-of-agent#ran","syntology_url":"https://syntology.ai/paper/2005.08874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08874"}},"official":{"repos":["HuTobias/HIGHLIGHTS-LRP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/lifelong-control-of-off-grid-microgrid-with","slug":"lifelong-control-of-off-grid-microgrid-with","title":"Lifelong Control of Off-grid Microgrid with Model Based Reinforcement Learning","date":"2020-05-16","arxiv_id":"2005.08006","repositories_listed":1,"syntology":null}],"record_sha256":"c0311592543638b35bc7ceb467b6d0a14ba1ab4fbe6c467d128c8ed29efa411d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}