{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/31","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":31,"pages_in_order":132,"rows_per_page":100,"rows":[3001,3100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/30","next":"/task/reinforcement-learning/papers/32","papers":[{"url":"/paper/obstacle-avoidance-and-navigation-utilizing","slug":"obstacle-avoidance-and-navigation-utilizing","title":"Obstacle Avoidance and Navigation Utilizing Reinforcement Learning with Reward Shaping","date":"2020-03-28","arxiv_id":"2003.12863","repositories_listed":1,"syntology":null},{"url":"/paper/policy-teaching-via-environment-poisoning","slug":"policy-teaching-via-environment-poisoning","title":"Policy Teaching via Environment Poisoning: Training-time Adversarial Attacks against Reinforcement Learning","date":"2020-03-28","arxiv_id":"2003.12909","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-teaching-via-environment-poisoning#ran","syntology_url":"https://syntology.ai/paper/2003.12909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12909"}},"official":{"repos":["adishs/icml2020_rl-policy-teaching_code"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-learning-in-asset-management-part-2","slug":"machine-learning-in-asset-management-part-2","title":"Machine Learning in Asset Management—Part 2: Portfolio Construction—Weight Optimization. The Journal of Financial Data Science","date":"2020-03-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fiber-a-platform-for-efficient-development","slug":"fiber-a-platform-for-efficient-development","title":"Fiber: A Platform for Efficient Development and Distributed Training for Reinforcement Learning and Population-Based Methods","date":"2020-03-25","arxiv_id":"2003.11164","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fiber-a-platform-for-efficient-development#ran","syntology_url":"https://syntology.ai/paper/2003.11164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11164"}},"official":null}},{"url":"/paper/an-empirical-investigation-of-the-challenges","slug":"an-empirical-investigation-of-the-challenges","title":"An empirical investigation of the challenges of real-world reinforcement learning","date":"2020-03-24","arxiv_id":"2003.11881","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-empirical-investigation-of-the-challenges#ran","syntology_url":"https://syntology.ai/paper/2003.11881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11881"}},"official":{"repos":["google-research/realworldrl_suite"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pads-policy-adapted-sampling-for-visual","slug":"pads-policy-adapted-sampling-for-visual","title":"PADS: Policy-Adapted Sampling for Visual Similarity Learning","date":"2020-03-24","arxiv_id":"2003.11113","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pads-policy-adapted-sampling-for-visual#ran","syntology_url":"https://syntology.ai/paper/2003.11113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11113"}},"official":{"repos":["Confusezius/CVPR2020_PADS"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-population-curriculum-for-1","slug":"evolutionary-population-curriculum-for-1","title":"Evolutionary Population Curriculum for Scaling Multi-Agent Reinforcement Learning","date":"2020-03-23","arxiv_id":"2003.10423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolutionary-population-curriculum-for-1#ran","syntology_url":"https://syntology.ai/paper/2003.10423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10423"}},"official":{"repos":["qian18long/epciclr2020"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-game-engine-accurate-learning","slug":"neural-game-engine-accurate-learning","title":"Neural Game Engine: Accurate learning of generalizable forward models from pixels","date":"2020-03-23","arxiv_id":"2003.10520","repositories_listed":1,"syntology":null},{"url":"/paper/using-deep-reinforcement-learning-methods-for","slug":"using-deep-reinforcement-learning-methods-for","title":"Using Deep Reinforcement Learning Methods for Autonomous Vessels in 2D Environments","date":"2020-03-23","arxiv_id":"2003.10249","repositories_listed":1,"syntology":null},{"url":"/paper/who2com-collaborative-perception-via","slug":"who2com-collaborative-perception-via","title":"Who2com: Collaborative Perception via Learnable Handshake Communication","date":"2020-03-21","arxiv_id":"2003.09575","repositories_listed":1,"syntology":null},{"url":"/paper/l2b-learning-to-balance-the-safety-efficiency","slug":"l2b-learning-to-balance-the-safety-efficiency","title":"L2B: Learning to Balance the Safety-Efficiency Trade-off in Interactive Crowd-aware Robot Navigation","date":"2020-03-20","arxiv_id":"2003.09207","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-of-control-affine","slug":"safe-reinforcement-learning-of-control-affine","title":"Safe Reinforcement Learning of Control-Affine Systems with Vertex Networks","date":"2020-03-20","arxiv_id":"2003.09488","repositories_listed":1,"syntology":null},{"url":"/paper/adjust-planning-strategies-to-accommodate","slug":"adjust-planning-strategies-to-accommodate","title":"Adjust Planning Strategies to Accommodate Reinforcement Learning Agents","date":"2020-03-19","arxiv_id":"2003.08554","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-poet-open-ended-reinforcement","slug":"enhanced-poet-open-ended-reinforcement","title":"Enhanced POET: Open-Ended Reinforcement Learning through Unbounded Invention of Learning Challenges and their Solutions","date":"2020-03-19","arxiv_id":"2003.08536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhanced-poet-open-ended-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.08536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08536"}},"official":{"repos":["uber-research/poet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-fly-via-deep-model-based","slug":"learning-to-fly-via-deep-model-based","title":"Learning to Fly via Deep Model-Based Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08876","repositories_listed":1,"syntology":null},{"url":"/paper/monotonic-value-function-factorisation-for","slug":"monotonic-value-function-factorisation-for","title":"Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monotonic-value-function-factorisation-for#ran","syntology_url":"https://syntology.ai/paper/2003.08839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08839"}},"official":{"repos":["oxwhirl/pymarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sapien-a-simulated-part-based-interactive","slug":"sapien-a-simulated-part-based-interactive","title":"SAPIEN: A SimulAted Part-based Interactive ENvironment","date":"2020-03-19","arxiv_id":"2003.08515","repositories_listed":1,"syntology":null},{"url":"/paper/social-navigation-with-human-empowerment","slug":"social-navigation-with-human-empowerment","title":"Social Navigation with Human Empowerment driven Deep Reinforcement Learning","date":"2020-03-18","arxiv_id":"2003.08158","repositories_listed":1,"syntology":null},{"url":"/paper/giving-up-control-neurons-as-reinforcement","slug":"giving-up-control-neurons-as-reinforcement","title":"Giving Up Control: Neurons as Reinforcement Learning Agents","date":"2020-03-17","arxiv_id":"2003.11642","repositories_listed":1,"syntology":null},{"url":"/paper/simultaneous-navigation-and-radio-mapping-for","slug":"simultaneous-navigation-and-radio-mapping-for","title":"Simultaneous Navigation and Radio Mapping for Cellular-Connected UAV with Deep Reinforcement Learning","date":"2020-03-17","arxiv_id":"2003.07574","repositories_listed":1,"syntology":null},{"url":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-discovering-of-causal","slug":"self-supervised-discovering-of-causal","title":"Self-Supervised Discovering of Interpretable Features for Reinforcement Learning","date":"2020-03-16","arxiv_id":"2003.07069","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-discovering-of-causal#ran","syntology_url":"https://syntology.ai/paper/2003.07069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07069"}},"official":{"repos":["shiwj16/SSINet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-efficient-exploration-for-rl-with","slug":"provably-efficient-exploration-for-rl-with","title":"Provably Efficient Exploration for Reinforcement Learning Using Unsupervised Learning","date":"2020-03-15","arxiv_id":"2003.06898","repositories_listed":1,"syntology":null},{"url":"/paper/target-driven-visual-navigation-exploiting","slug":"target-driven-visual-navigation-exploiting","title":"Learning hierarchical relationships for object-goal navigation","date":"2020-03-15","arxiv_id":"2003.06749","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/target-driven-visual-navigation-exploiting#ran","syntology_url":"https://syntology.ai/paper/2003.06749","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06749"}},"official":null}},{"url":"/paper/learning-reinforced-agents-with","slug":"learning-reinforced-agents-with","title":"Towards Causality-Aware Inferring: A Sequential Discriminative Approach for Medical Diagnosis","date":"2020-03-14","arxiv_id":"2003.06534","repositories_listed":1,"syntology":null},{"url":"/paper/deep-deterministic-portfolio-optimization","slug":"deep-deterministic-portfolio-optimization","title":"Deep Deterministic Portfolio Optimization","date":"2020-03-13","arxiv_id":"2003.06497","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-deterministic-portfolio-optimization#ran","syntology_url":"https://syntology.ai/paper/2003.06497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06497"}},"official":{"repos":["CFMTech/Deep-RL-for-Portfolio-Optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-trojai-software-framework-an-opensource","slug":"the-trojai-software-framework-an-opensource","title":"The TrojAI Software Framework: An OpenSource tool for Embedding Trojans into Deep Learning Models","date":"2020-03-13","arxiv_id":"2003.07233","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-trojai-software-framework-an-opensource#ran","syntology_url":"https://syntology.ai/paper/2003.07233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07233"}},"official":null}},{"url":"/paper/invariant-causal-prediction-for-block-mdps","slug":"invariant-causal-prediction-for-block-mdps","title":"Invariant Causal Prediction for Block MDPs","date":"2020-03-12","arxiv_id":"2003.06016","repositories_listed":1,"syntology":null},{"url":"/paper/option-discovery-in-the-absence-of-rewards","slug":"option-discovery-in-the-absence-of-rewards","title":"Option Discovery in the Absence of Rewards with Manifold Analysis","date":"2020-03-12","arxiv_id":"2003.05878","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-negative-sampling-over-knowledge","slug":"reinforced-negative-sampling-over-knowledge","title":"Reinforced Negative Sampling over Knowledge Graph for Recommendation","date":"2020-03-12","arxiv_id":"2003.05753","repositories_listed":1,"syntology":null},{"url":"/paper/the-chefs-hat-simulation-environment-for","slug":"the-chefs-hat-simulation-environment-for","title":"The Chef's Hat Simulation Environment for Reinforcement-Learning-Based Agents","date":"2020-03-12","arxiv_id":"2003.05861","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-curiosity-algorithms-1","slug":"meta-learning-curiosity-algorithms-1","title":"Meta-learning curiosity algorithms","date":"2020-03-11","arxiv_id":"2003.05325","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-learning-curiosity-algorithms-1#ran","syntology_url":"https://syntology.ai/paper/2003.05325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05325"}},"official":{"repos":["mfranzs/meta-learning-curiosity-algorithms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiplicative-controller-fusion-a-hybrid","slug":"multiplicative-controller-fusion-a-hybrid","title":"Multiplicative Controller Fusion: Leveraging Algorithmic Priors for Sample-efficient Reinforcement Learning and Safe Sim-To-Real Transfer","date":"2020-03-11","arxiv_id":"2003.05117","repositories_listed":1,"syntology":null},{"url":"/paper/online-meta-critic-learning-for-off-policy-1","slug":"online-meta-critic-learning-for-off-policy-1","title":"Online Meta-Critic Learning for Off-Policy Actor-Critic Methods","date":"2020-03-11","arxiv_id":"2003.05334","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/online-meta-critic-learning-for-off-policy-1#ran","syntology_url":"https://syntology.ai/paper/2003.05334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05334"}},"official":{"repos":["zwfightzw/Meta-Critic"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/explore-and-exploit-with-heterotic-line","slug":"explore-and-exploit-with-heterotic-line","title":"Explore and Exploit with Heterotic Line Bundle Models","date":"2020-03-10","arxiv_id":"2003.04817","repositories_listed":1,"syntology":null},{"url":"/paper/fast-online-adaptation-in-robotics-through","slug":"fast-online-adaptation-in-robotics-through","title":"Fast Online Adaptation in Robotics through Meta-Learning Embeddings of Simulated Priors","date":"2020-03-10","arxiv_id":"2003.04663","repositories_listed":1,"syntology":null},{"url":"/paper/learning-discrete-state-abstractions-with","slug":"learning-discrete-state-abstractions-with","title":"Learning Discrete State Abstractions With Deep Variational Inference","date":"2020-03-09","arxiv_id":"2003.04300","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-discrete-state-abstractions-with#ran","syntology_url":"https://syntology.ai/paper/2003.04300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04300"}},"official":{"repos":["ondrejba/discrete_abstractions"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/stable-policy-optimization-via-off-policy","slug":"stable-policy-optimization-via-off-policy","title":"Stable Policy Optimization via Off-Policy Divergence Regularization","date":"2020-03-09","arxiv_id":"2003.04108","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stable-policy-optimization-via-off-policy#ran","syntology_url":"https://syntology.ai/paper/2003.04108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04108"}},"official":{"repos":["facebookresearch/ppo-dice"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dada-differentiable-automatic-data","slug":"dada-differentiable-automatic-data","title":"DADA: Differentiable Automatic Data Augmentation","date":"2020-03-08","arxiv_id":"2003.03780","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dada-differentiable-automatic-data#ran","syntology_url":"https://syntology.ai/paper/2003.03780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.03780"}},"official":{"repos":["VDIGPKU/DADA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-robustness-of-cooperative-multi-agent","slug":"on-the-robustness-of-cooperative-multi-agent","title":"On the Robustness of Cooperative Multi-Agent Reinforcement Learning","date":"2020-03-08","arxiv_id":"2003.03722","repositories_listed":1,"syntology":null},{"url":"/paper/ig-rl-inductive-graph-reinforcement-learning","slug":"ig-rl-inductive-graph-reinforcement-learning","title":"IG-RL: Inductive Graph Reinforcement Learning for Massive-Scale Traffic Signal Control","date":"2020-03-06","arxiv_id":"2003.05738","repositories_listed":1,"syntology":null},{"url":"/paper/learning-view-and-target-invariant-visual","slug":"learning-view-and-target-invariant-visual","title":"Learning View and Target Invariant Visual Servoing for Navigation","date":"2020-03-04","arxiv_id":"2003.02327","repositories_listed":1,"syntology":null},{"url":"/paper/can-increasing-input-dimensionality-improve","slug":"can-increasing-input-dimensionality-improve","title":"Can Increasing Input Dimensionality Improve Deep Reinforcement Learning?","date":"2020-03-03","arxiv_id":"2003.01629","repositories_listed":1,"syntology":null},{"url":"/paper/contention-window-optimization-in-ieee","slug":"contention-window-optimization-in-ieee","title":"Contention Window Optimization in IEEE 802.11ax Networks with Deep Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01492","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-synaptic-plasticity-with-online","slug":"embodied-synaptic-plasticity-with-online","title":"Embodied Synaptic Plasticity with Online Reinforcement learning","date":"2020-03-03","arxiv_id":"2003.01431","repositories_listed":1,"syntology":null},{"url":"/paper/robust-market-making-via-adversarial","slug":"robust-market-making-via-adversarial","title":"Robust Market Making via Adversarial Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01820","repositories_listed":1,"syntology":null},{"url":"/paper/autophase-juggling-hls-phase-orderings-in","slug":"autophase-juggling-hls-phase-orderings-in","title":"AutoPhase: Juggling HLS Phase Orderings in Random Forests with Deep Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00671","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autophase-juggling-hls-phase-orderings-in#ran","syntology_url":"https://syntology.ai/paper/2003.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00671"}},"official":{"repos":["ucb-bar/autophase"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-easy-to-complex-adaptive-multi","slug":"learning-from-easy-to-complex-adaptive-multi","title":"Learning from Easy to Complex: Adaptive Multi-curricula Learning for Neural Dialogue Generation","date":"2020-03-02","arxiv_id":"2003.00639","repositories_listed":1,"syntology":null},{"url":"/paper/mvp-unified-motion-and-visual-self-supervised","slug":"mvp-unified-motion-and-visual-self-supervised","title":"MVP: Unified Motion and Visual Self-Supervised Learning for Large-Scale Robotic Navigation","date":"2020-03-02","arxiv_id":"2003.00667","repositories_listed":1,"syntology":null},{"url":"/paper/ppmc-training-algorithm-a-robot-independent","slug":"ppmc-training-algorithm-a-robot-independent","title":"PPMC RL Training Algorithm: Rough Terrain Intelligent Robots through Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.02655","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-co-learning-of-deep-and-spiking","slug":"reinforcement-co-learning-of-deep-and-spiking","title":"Reinforcement co-Learning of Deep and Spiking Neural Networks for Energy-Efficient Mapless Navigation with Neuromorphic Hardware","date":"2020-03-02","arxiv_id":"2003.01157","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-co-learning-of-deep-and-spiking#ran","syntology_url":"https://syntology.ai/paper/2003.01157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.01157"}},"official":{"repos":["combra-lab/spiking-ddpg-mapless-navigation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hybrid-stochastic-policy-gradient-algorithm","slug":"a-hybrid-stochastic-policy-gradient-algorithm","title":"A Hybrid Stochastic Policy Gradient Algorithm for Reinforcement Learning","date":"2020-03-01","arxiv_id":"2003.00430","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hybrid-stochastic-policy-gradient-algorithm#ran","syntology_url":"https://syntology.ai/paper/2003.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00430"}},"official":{"repos":["unc-optimization/ProxHSPGA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-catastrophic-interference-in-atari-2600","slug":"on-catastrophic-interference-in-atari-2600","title":"On Catastrophic Interference in Atari 2600 Games","date":"2020-02-28","arxiv_id":"2002.12499","repositories_listed":1,"syntology":null},{"url":"/paper/policy-aware-model-learning-for-policy","slug":"policy-aware-model-learning-for-policy","title":"Policy-Aware Model Learning for Policy Gradient Methods","date":"2020-02-28","arxiv_id":"2003.00030","repositories_listed":1,"syntology":null},{"url":"/paper/probably-approximately-correct-vision-based","slug":"probably-approximately-correct-vision-based","title":"Probably Approximately Correct Vision-Based Planning using Motion Primitives","date":"2020-02-28","arxiv_id":"2002.12852","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-robotic-nanofabrication-with","slug":"autonomous-robotic-nanofabrication-with","title":"Autonomous robotic nanofabrication with reinforcement learning","date":"2020-02-27","arxiv_id":"2002.11952","repositories_listed":1,"syntology":null},{"url":"/paper/gamma-reward-a-novel-multi-agent","slug":"gamma-reward-a-novel-multi-agent","title":"Learning Scalable Multi-Agent Coordination by Spatial Differentiation for Traffic Signal Control","date":"2020-02-27","arxiv_id":"2002.11874","repositories_listed":1,"syntology":null},{"url":"/paper/plannable-approximations-to-mdp-homomorphisms","slug":"plannable-approximations-to-mdp-homomorphisms","title":"Plannable Approximations to MDP Homomorphisms: Equivariance under Actions","date":"2020-02-27","arxiv_id":"2002.11963","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/plannable-approximations-to-mdp-homomorphisms#ran","syntology_url":"https://syntology.ai/paper/2002.11963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.11963"}},"official":{"repos":["ElisevanderPol/prae"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-of-risk-constrained","slug":"reinforcement-learning-of-risk-constrained","title":"Reinforcement Learning of Risk-Constrained Policies in Markov Decision Processes","date":"2020-02-27","arxiv_id":"2002.12086","repositories_listed":1,"syntology":null},{"url":"/paper/training-adversarial-agents-to-exploit","slug":"training-adversarial-agents-to-exploit","title":"Training Adversarial Agents to Exploit Weaknesses in Deep Control Policies","date":"2020-02-27","arxiv_id":"2002.12078","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-control-for","slug":"efficient-reinforcement-learning-control-for","title":"Efficient reinforcement learning control for continuum robots based on Inexplicit Prior Knowledge","date":"2020-02-26","arxiv_id":"2002.11573","repositories_listed":1,"syntology":null},{"url":"/paper/mid-flight-propeller-failure-detection-and","slug":"mid-flight-propeller-failure-detection-and","title":"Mid-flight Propeller Failure Detection and Control of Propeller-deficient Quadcopter using Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11564","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-exploration-even-with-a-1#ran","syntology_url":"https://syntology.ai/paper/2002.12174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.12174"}},"official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-reinforcement-learning-in-the","slug":"using-reinforcement-learning-in-the","title":"Using Reinforcement Learning in the Algorithmic Trading Problem","date":"2020-02-26","arxiv_id":"2002.11523","repositories_listed":1,"syntology":null},{"url":"/paper/human-apprenticeship-learning-via-kernel","slug":"human-apprenticeship-learning-via-kernel","title":"Reward Shaping for Human Learning via Inverse Reinforcement Learning","date":"2020-02-25","arxiv_id":"2002.10904","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-deep-reinforcement-learning-with","slug":"off-policy-deep-reinforcement-learning-with","title":"Off-Policy Deep Reinforcement Learning with Analogous Disentangled Exploration","date":"2020-02-25","arxiv_id":"2002.10738","repositories_listed":1,"syntology":null},{"url":"/paper/rewriting-history-with-inverse-rl-hindsight","slug":"rewriting-history-with-inverse-rl-hindsight","title":"Rewriting History with Inverse RL: Hindsight Inference for Policy Improvement","date":"2020-02-25","arxiv_id":"2002.11089","repositories_listed":1,"syntology":null},{"url":"/paper/whole-body-control-of-a-mobile-manipulator","slug":"whole-body-control-of-a-mobile-manipulator","title":"Whole-Body Control of a Mobile Manipulator using End-to-End Reinforcement Learning","date":"2020-02-25","arxiv_id":"2003.02637","repositories_listed":1,"syntology":null},{"url":"/paper/reconfigurable-intelligent-surface-assisted","slug":"reconfigurable-intelligent-surface-assisted","title":"Reconfigurable Intelligent Surface Assisted Multiuser MISO Systems Exploiting Deep Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10072","repositories_listed":1,"syntology":null},{"url":"/paper/robust-learning-based-control-via","slug":"robust-learning-based-control-via","title":"Robust Learning-Based Control via Bootstrapped Multiplicative Noise","date":"2020-02-24","arxiv_id":"2002.10069","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-for-probabilistic","slug":"safe-reinforcement-learning-for-probabilistic","title":"Safe reinforcement learning for probabilistic reachability and safety specifications: A Lyapunov-based approach","date":"2020-02-24","arxiv_id":"2002.10126","repositories_listed":1,"syntology":null},{"url":"/paper/sketch-less-for-more-on-the-fly-fine-grained","slug":"sketch-less-for-more-on-the-fly-fine-grained","title":"Sketch Less for More: On-the-Fly Fine-Grained Sketch Based Image Retrieval","date":"2020-02-24","arxiv_id":"2002.10310","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-particle-filter-reinforcement-1","slug":"discriminative-particle-filter-reinforcement-1","title":"Discriminative Particle Filter Reinforcement Learning for Complex Partial Observations","date":"2020-02-23","arxiv_id":"2002.09884","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-framework-for-deep","slug":"reinforcement-learning-framework-for-deep","title":"Reinforcement Learning Framework for Deep Brain Stimulation Study","date":"2020-02-22","arxiv_id":"2002.10948","repositories_listed":1,"syntology":null},{"url":"/paper/tuning-free-plug-and-play-proximal-algorithm","slug":"tuning-free-plug-and-play-proximal-algorithm","title":"Tuning-free Plug-and-Play Proximal Algorithm for Inverse Imaging Problems","date":"2020-02-22","arxiv_id":"2002.09611","repositories_listed":1,"syntology":null},{"url":"/paper/gendice-generalized-offline-estimation-of-1","slug":"gendice-generalized-offline-estimation-of-1","title":"GenDICE: Generalized Offline Estimation of Stationary Values","date":"2020-02-21","arxiv_id":"2002.09072","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-walk-in-the-real-world-with","slug":"learning-to-walk-in-the-real-world-with","title":"Learning to Walk in the Real World with Minimal Human Effort","date":"2020-02-20","arxiv_id":"2002.08550","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-through","slug":"efficient-deep-reinforcement-learning-through","title":"Efficient Deep Reinforcement Learning via Adaptive Policy Transfer","date":"2020-02-19","arxiv_id":"2002.08037","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-avoid-being-eaten-by-a-grue","slug":"how-to-avoid-being-eaten-by-a-grue","title":"How To Avoid Being Eaten By a Grue: Exploration Strategies for Text-Adventure Agents","date":"2020-02-19","arxiv_id":"2002.08795","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-avoid-being-eaten-by-a-grue#ran","syntology_url":"https://syntology.ai/paper/2002.08795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08795"}},"official":{"repos":["rajammanabrolu/Q-BERT"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/sim2real-transfer-for-reinforcement-learning","slug":"sim2real-transfer-for-reinforcement-learning","title":"Sim2Real Transfer for Reinforcement Learning without Dynamics Randomization","date":"2020-02-19","arxiv_id":"2002.11635","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-estimator-selection-for-off-policy","slug":"adaptive-estimator-selection-for-off-policy","title":"Adaptive Estimator Selection for Off-Policy Evaluation","date":"2020-02-18","arxiv_id":"2002.07729","repositories_listed":1,"syntology":null},{"url":"/paper/generating-automatic-curricula-via-self","slug":"generating-automatic-curricula-via-self","title":"Generating Automatic Curricula via Self-Supervised Active Domain Randomization","date":"2020-02-18","arxiv_id":"2002.07911","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-automatic-curricula-via-self#ran","syntology_url":"https://syntology.ai/paper/2002.07911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.07911"}},"official":null}},{"url":"/paper/reinforcement-learning-for-molecular-design","slug":"reinforcement-learning-for-molecular-design","title":"Reinforcement Learning for Molecular Design Guided by Quantum Mechanics","date":"2020-02-18","arxiv_id":"2002.07717","repositories_listed":1,"syntology":null},{"url":"/paper/spatial-concept-based-navigation-with-human","slug":"spatial-concept-based-navigation-with-human","title":"Spatial Concept-Based Navigation with Human Speech Instructions via Probabilistic Inference on Bayesian Generative Model","date":"2020-02-18","arxiv_id":"2002.07381","repositories_listed":1,"syntology":null},{"url":"/paper/control-frequency-adaptation-via-action","slug":"control-frequency-adaptation-via-action","title":"Control Frequency Adaptation via Action Persistence in Batch Reinforcement Learning","date":"2020-02-17","arxiv_id":"2002.06836","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/control-frequency-adaptation-via-action#ran","syntology_url":"https://syntology.ai/paper/2002.06836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06836"}},"official":{"repos":["albertometelli/pfqi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kalman-meets-bellman-improving-policy","slug":"kalman-meets-bellman-improving-policy","title":"Kalman meets Bellman: Improving Policy Evaluation through Value Tracking","date":"2020-02-17","arxiv_id":"2002.07171","repositories_listed":1,"syntology":null},{"url":"/paper/r-maddpg-for-partially-observable","slug":"r-maddpg-for-partially-observable","title":"R-MADDPG for Partially Observable Environments and Limited Communication","date":"2020-02-16","arxiv_id":"2002.06684","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-active-learning-for-image-1","slug":"reinforced-active-learning-for-image-1","title":"Reinforced active learning for image segmentation","date":"2020-02-16","arxiv_id":"2002.06583","repositories_listed":1,"syntology":null},{"url":"/paper/deep-rl-agent-for-a-real-time-action-strategy","slug":"deep-rl-agent-for-a-real-time-action-strategy","title":"Deep RL Agent for a Real-Time Action Strategy Game","date":"2020-02-15","arxiv_id":"2002.06290","repositories_listed":1,"syntology":null},{"url":"/paper/loop-estimator-for-discounted-values-in","slug":"loop-estimator-for-discounted-values-in","title":"Loop Estimator for Discounted Values in Markov Reward Processes","date":"2020-02-15","arxiv_id":"2002.06299","repositories_listed":1,"syntology":null},{"url":"/paper/universal-value-density-estimation-for","slug":"universal-value-density-estimation-for","title":"Universal Value Density Estimation for Imitation Learning and Goal-Conditioned Reinforcement Learning","date":"2020-02-15","arxiv_id":"2002.06473","repositories_listed":1,"syntology":null},{"url":"/paper/extended-markov-games-to-learn-multiple-tasks","slug":"extended-markov-games-to-learn-multiple-tasks","title":"Extended Markov Games to Learn Multiple Tasks in Multi-Agent Reinforcement Learning","date":"2020-02-14","arxiv_id":"2002.06000","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-via-adversarial-1","slug":"robust-reinforcement-learning-via-adversarial-1","title":"Robust Reinforcement Learning via Adversarial training with Langevin Dynamics","date":"2020-02-14","arxiv_id":"2002.06063","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-end-to-end-learning-on","slug":"a-framework-for-end-to-end-learning-on","title":"A Framework for End-to-End Learning on Semantic Tree-Structured Data","date":"2020-02-13","arxiv_id":"2002.05707","repositories_listed":1,"syntology":null},{"url":"/paper/effective-reinforcement-learning-through","slug":"effective-reinforcement-learning-through","title":"Effective Reinforcement Learning through Evolutionary Surrogate-Assisted Prescription","date":"2020-02-13","arxiv_id":"2002.05368","repositories_listed":1,"syntology":null},{"url":"/paper/hoplite-efficient-collective-communication","slug":"hoplite-efficient-collective-communication","title":"Hoplite: Efficient and Fault-Tolerant Collective Communication for Task-Based Distributed Systems","date":"2020-02-13","arxiv_id":"2002.05814","repositories_listed":1,"syntology":null},{"url":"/paper/fully-differentiable-procedural-content","slug":"fully-differentiable-procedural-content","title":"Learning to Generate Levels From Nothing","date":"2020-02-12","arxiv_id":"2002.05259","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fully-differentiable-procedural-content#ran","syntology_url":"https://syntology.ai/paper/2002.05259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05259"}},"official":{"repos":["pbontrager/GenerativePlayingNetworks"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-convergent-policy-gradient-methods","slug":"provably-convergent-policy-gradient-methods","title":"On the Convergence Theory of Debiased Model-Agnostic Meta-Reinforcement Learning","date":"2020-02-12","arxiv_id":"2002.05135","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-quantum","slug":"reinforcement-learning-enhanced-quantum","title":"Reinforcement Learning Enhanced Quantum-inspired Algorithm for Combinatorial Optimization","date":"2020-02-11","arxiv_id":"2002.04676","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-enhanced-quantum#ran","syntology_url":"https://syntology.ai/paper/2002.04676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.04676"}},"official":{"repos":["BeloborodovDS/SIMCIM-RL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/discrete-action-on-policy-learning-with","slug":"discrete-action-on-policy-learning-with","title":"Discrete Action On-Policy Learning with Action-Value Critic","date":"2020-02-10","arxiv_id":"2002.03534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-action-on-policy-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.03534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03534"}},"official":{"repos":["yuguangyue/CARSM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"98e6c46cc1d3ef9eff51108e77415e0e691167dfa15e248561c698d5f75ed45a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}