{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/32","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":32,"pages_in_order":135,"rows_per_page":100,"rows":[3101,3200],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/31","next":"/task/reinforcement-learning-2/papers/33","papers":[{"url":"/paper/learning-associative-inference-using-fast-1","slug":"learning-associative-inference-using-fast-1","title":"Learning Associative Inference Using Fast Weight Memory","date":"2020-11-16","arxiv_id":"2011.07831","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-associative-inference-using-fast-1#ran","syntology_url":"https://syntology.ai/paper/2011.07831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07831"}},"official":{"repos":["ischlag/Fast-Weight-Memory-public"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-reinforcement-learning-policies-for","slug":"scalable-reinforcement-learning-policies-for","title":"Scalable Reinforcement Learning Policies for Multi-Agent Control","date":"2020-11-16","arxiv_id":"2011.08055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-reinforcement-learning-policies-for#ran","syntology_url":"https://syntology.ai/paper/2011.08055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.08055"}},"official":{"repos":["christopher-hsu/scalableMARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cdt-cascading-decision-trees-for-explainable-1","slug":"cdt-cascading-decision-trees-for-explainable-1","title":"CDT: Cascading Decision Trees for Explainable Reinforcement Learning","date":"2020-11-15","arxiv_id":"2011.07553","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-cybersecurity","slug":"deep-reinforcement-learning-for-cybersecurity","title":"Deep Reinforcement Learning for Cybersecurity Assessment of Wind Integrated Power Systems","date":"2020-11-15","arxiv_id":"2007.03025","repositories_listed":1,"syntology":null},{"url":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tonic-a-deep-reinforcement-learning-library#ran","syntology_url":"https://syntology.ai/paper/2011.07537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07537"}},"official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deepmind-lab2d","slug":"deepmind-lab2d","title":"DeepMind Lab2D","date":"2020-11-13","arxiv_id":"2011.07027","repositories_listed":1,"syntology":null},{"url":"/paper/query-based-targeted-action-space-adversarial","slug":"query-based-targeted-action-space-adversarial","title":"Query-based Targeted Action-Space Adversarial Policies on Deep Reinforcement Learning Agents","date":"2020-11-13","arxiv_id":"2011.07114","repositories_listed":1,"syntology":null},{"url":"/paper/roll-visual-self-supervised-reinforcement","slug":"roll-visual-self-supervised-reinforcement","title":"ROLL: Visual Self-Supervised Reinforcement Learning with Object Reasoning","date":"2020-11-13","arxiv_id":"2011.06777","repositories_listed":1,"syntology":null},{"url":"/paper/gaussian-ram-lightweight-image-classification","slug":"gaussian-ram-lightweight-image-classification","title":"Gaussian RAM: Lightweight Image Classification via Stochastic Retina-Inspired Glimpse and Reinforcement Learning","date":"2020-11-12","arxiv_id":"2011.06190","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-large-scale-fleet-management-on-a","slug":"optimizing-large-scale-fleet-management-on-a","title":"Optimizing Large-Scale Fleet Management on a Road Network using Multi-Agent Deep Reinforcement Learning with Graph Neural Network","date":"2020-11-12","arxiv_id":"2011.06175","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-videos-combining","slug":"reinforcement-learning-with-videos-combining","title":"Reinforcement Learning with Videos: Combining Offline Observations with Interaction","date":"2020-11-12","arxiv_id":"2011.06507","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-motion-planning-for-multi-robot","slug":"decentralized-motion-planning-for-multi-robot","title":"Decentralized Motion Planning for Multi-Robot Navigation using Deep Reinforcement Learning","date":"2020-11-11","arxiv_id":"2011.05605","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-experiments-and","slug":"reinforcement-learning-experiments-and","title":"Reinforcement Learning Experiments and Benchmark for Solving Robotic Reaching Tasks","date":"2020-11-11","arxiv_id":"2011.05782","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-for-general","slug":"robust-reinforcement-learning-for-general","title":"Reinforcement Learning with Dual-Observation for General Video Game Playing","date":"2020-11-11","arxiv_id":"2011.05622","repositories_listed":1,"syntology":null},{"url":"/paper/f-irl-inverse-reinforcement-learning-via","slug":"f-irl-inverse-reinforcement-learning-via","title":"f-IRL: Inverse Reinforcement Learning via State Marginal Matching","date":"2020-11-09","arxiv_id":"2011.04709","repositories_listed":1,"syntology":null},{"url":"/paper/geometric-deep-reinforcement-learning-for","slug":"geometric-deep-reinforcement-learning-for","title":"Geometric Deep Reinforcement Learning for Dynamic DAG Scheduling","date":"2020-11-09","arxiv_id":"2011.04333","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-planning-for-autonomous-vehicles","slug":"trajectory-planning-for-autonomous-vehicles","title":"Trajectory Planning for Autonomous Vehicles Using Hierarchical Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04752","repositories_listed":1,"syntology":null},{"url":"/paper/drafting-in-collectible-card-games-via","slug":"drafting-in-collectible-card-games-via","title":"Drafting in Collectible Card Games via Reinforcement Learning","date":"2020-11-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-a-decentralized-multi-arm-motion","slug":"learning-a-decentralized-multi-arm-motion","title":"Learning a Decentralized Multi-arm Motion Planner","date":"2020-11-05","arxiv_id":"2011.02608","repositories_listed":1,"syntology":null},{"url":"/paper/realant-an-open-source-low-cost-quadruped-for","slug":"realant-an-open-source-low-cost-quadruped-for","title":"RealAnt: An Open-Source Low-Cost Quadruped for Education and Research in Real-World Reinforcement Learning","date":"2020-11-05","arxiv_id":"2011.03085","repositories_listed":1,"syntology":null},{"url":"/paper/learning-trajectories-for-visual-inertial","slug":"learning-trajectories-for-visual-inertial","title":"Learning Trajectories for Visual-Inertial System Calibration via Model-based Heuristic Deep Reinforcement Learning","date":"2020-11-04","arxiv_id":"2011.02574","repositories_listed":1,"syntology":null},{"url":"/paper/xcsf-for-automatic-test-case-prioritization","slug":"xcsf-for-automatic-test-case-prioritization","title":"XCSF for Automatic Test Case Prioritization","date":"2020-11-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/causal-campbell-goodhart-s-law-and","slug":"causal-campbell-goodhart-s-law-and","title":"Causal Campbell-Goodhart's law and Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01010","repositories_listed":1,"syntology":null},{"url":"/paper/instance-based-generalization-in","slug":"instance-based-generalization-in","title":"Instance based Generalization in Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01089","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instance-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2011.01089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01089"}},"official":{"repos":["MartinBertran/InstanceAgnosticPolicyEnsembles"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for","slug":"multi-agent-reinforcement-learning-for","title":"Multi-Agent Reinforcement Learning for Visibility-based Persistent Monitoring","date":"2020-11-02","arxiv_id":"2011.01129","repositories_listed":1,"syntology":null},{"url":"/paper/self-driving-network-and-service-coordination","slug":"self-driving-network-and-service-coordination","title":"Self-Driving Network and Service Coordination Using Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-overview-of-multi-agent-reinforcement","slug":"an-overview-of-multi-agent-reinforcement","title":"Game-Theoretic Multiagent Reinforcement Learning","date":"2020-11-01","arxiv_id":"2011.00583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-overview-of-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2011.00583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00583"}},"official":null}},{"url":"/paper/ask-your-humans-using-human-instructions-to-1","slug":"ask-your-humans-using-human-instructions-to-1","title":"Ask Your Humans: Using Human Instructions to Improve Generalization in Reinforcement Learning","date":"2020-11-01","arxiv_id":"2011.00517","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ask-your-humans-using-human-instructions-to-1#ran","syntology_url":"https://syntology.ai/paper/2011.00517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00517"}},"official":{"repos":["valeriechen/ask-your-humans"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/guided-dialogue-policy-learning-without","slug":"guided-dialogue-policy-learning-without","title":"Guided Dialogue Policy Learning without Adversarial Learning in the Loop","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-policy-gradient-algorithm-for-learning-to-1","slug":"a-policy-gradient-algorithm-for-learning-to-1","title":"A Policy Gradient Algorithm for Learning to Learn in Multiagent Reinforcement Learning","date":"2020-10-31","arxiv_id":"2011.00382","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-policy-gradient-algorithm-for-learning-to-1#ran","syntology_url":"https://syntology.ai/paper/2011.00382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00382"}},"official":{"repos":["dkkim93/meta-mapg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/firecommander-an-interactive-probabilistic","slug":"firecommander-an-interactive-probabilistic","title":"FireCommander: An Interactive, Probabilistic Multi-agent Environment for Heterogeneous Robot Teams","date":"2020-10-31","arxiv_id":"2011.00165","repositories_listed":1,"syntology":null},{"url":"/paper/pseudo-random-number-generation-through","slug":"pseudo-random-number-generation-through","title":"Pseudo Random Number Generation through Reinforcement Learning and Recurrent Neural Networks","date":"2020-10-31","arxiv_id":"2011.02909","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-complex-knowledge-base-question","slug":"few-shot-complex-knowledge-base-question","title":"Few-Shot Complex Knowledge Base Question Answering via Meta Reinforcement Learning","date":"2020-10-29","arxiv_id":"2010.15877","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/few-shot-complex-knowledge-base-question#ran","syntology_url":"https://syntology.ai/paper/2010.15877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.15877"}},"official":{"repos":["DevinJake/MRL-CQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/personalized-multimorbidity-management-for","slug":"personalized-multimorbidity-management-for","title":"Personalized Multimorbidity Management for Patients with Type 2 Diabetes Using Reinforcement Learning of Electronic Health Records","date":"2020-10-29","arxiv_id":"2011.02287","repositories_listed":1,"syntology":null},{"url":"/paper/cog-connecting-new-skills-to-past-experience","slug":"cog-connecting-new-skills-to-past-experience","title":"COG: Connecting New Skills to Past Experience with Offline Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14500","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-under-parameterization-inhibits-data-1","slug":"implicit-under-parameterization-inhibits-data-1","title":"Implicit Under-Parameterization Inhibits Data-Efficient Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-under-parameterization-inhibits-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.14498"}},"official":null}},{"url":"/paper/improving-reinforcement-learning-for-neural","slug":"improving-reinforcement-learning-for-neural","title":"RH-Net: Improving Neural Relation Extraction via Reinforcement Learning and Hierarchical Relational Searching","date":"2020-10-27","arxiv_id":"2010.14255","repositories_listed":1,"syntology":null},{"url":"/paper/learning-financial-asset-specific-trading","slug":"learning-financial-asset-specific-trading","title":"Learning Financial Asset-Specific Trading Rules via Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14194","repositories_listed":1,"syntology":null},{"url":"/paper/meld-meta-reinforcement-learning-from-images","slug":"meld-meta-reinforcement-learning-from-images","title":"MELD: Meta-Reinforcement Learning from Images via Latent State Models","date":"2020-10-26","arxiv_id":"2010.13957","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-wise-multiple-choice-learning-for","slug":"trajectory-wise-multiple-choice-learning-for","title":"Trajectory-wise Multiple Choice Learning for Dynamics Generalization in Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13303","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-wise-multiple-choice-learning-for#ran","syntology_url":"https://syntology.ai/paper/2010.13303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13303"}},"official":{"repos":["younggyoseo/trajectory_mcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-make-deep-rl-work-in-practice","slug":"how-to-make-deep-rl-work-in-practice","title":"How to Make Deep RL Work in Practice","date":"2020-10-25","arxiv_id":"2010.13083","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-deceive-knowledge-graph-augmented-1","slug":"learning-to-deceive-knowledge-graph-augmented-1","title":"Learning to Deceive Knowledge Graph Augmented Models via Targeted Perturbation","date":"2020-10-24","arxiv_id":"2010.12872","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-imagination-and-reality-for-model","slug":"bridging-imagination-and-reality-for-model","title":"Bridging Imagination and Reality for Model-Based Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12142","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-imagination-and-reality-for-model#ran","syntology_url":"https://syntology.ai/paper/2010.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12142"}},"official":{"repos":["Mehooz/BIRD_code"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/multi-uav-path-planning-for-wireless-data","slug":"multi-uav-path-planning-for-wireless-data","title":"Multi-UAV Path Planning for Wireless Data Harvesting with Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12461","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-policy-improvement-for-non","slug":"towards-safe-policy-improvement-for-non","title":"Towards Safe Policy Improvement for Non-Stationary MDPs","date":"2020-10-23","arxiv_id":"2010.12645","repositories_listed":1,"syntology":null},{"url":"/paper/batch-exploration-with-examples-for-scalable","slug":"batch-exploration-with-examples-for-scalable","title":"Batch Exploration with Examples for Scalable Robotic Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11917","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-stacked","slug":"deep-reinforcement-learning-with-stacked","title":"Deep Reinforcement Learning with Stacked Hierarchical Attention for Text-based Games","date":"2020-10-22","arxiv_id":"2010.11655","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-with-stacked#ran","syntology_url":"https://syntology.ai/paper/2010.11655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11655"}},"official":{"repos":["YunqiuXu/SHA-KG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/drift-detection-in-episodic-data-detect-when-1","slug":"drift-detection-in-episodic-data-detect-when-1","title":"Detecting Rewards Deterioration in Episodic Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11660","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drift-detection-in-episodic-data-detect-when-1#ran","syntology_url":"https://syntology.ai/paper/2010.11660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11660"}},"official":{"repos":["ido90/Rewards-Deterioration-Detection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-combinatorial","slug":"reinforcement-learning-with-combinatorial","title":"Reinforcement Learning with Combinatorial Actions: An Application to Vehicle Routing","date":"2020-10-22","arxiv_id":"2010.12001","repositories_listed":1,"syntology":null},{"url":"/paper/parenting-via-model-agnostic-reinforcement","slug":"parenting-via-model-agnostic-reinforcement","title":"PARENTing via Model-Agnostic Reinforcement Learning to Correct Pathological Behaviors in Data-to-Text Generation","date":"2020-10-21","arxiv_id":"2010.10866","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-amortized-policy-optimization-1","slug":"iterative-amortized-policy-optimization-1","title":"Iterative Amortized Policy Optimization","date":"2020-10-20","arxiv_id":"2010.10670","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/iterative-amortized-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/2010.10670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10670"}},"official":{"repos":["joelouismarino/variational_rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reinforcement-learning-for-optimization-of","slug":"reinforcement-learning-for-optimization-of","title":"Reinforcement Learning for Optimization of COVID-19 Mitigation policies","date":"2020-10-20","arxiv_id":"2010.10560","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-for-optimization-of#ran","syntology_url":"https://syntology.ai/paper/2010.10560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10560"}},"official":{"repos":["SonyAI/PandemicSimulator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-with-population","slug":"deep-reinforcement-learning-with-population","title":"Deep Reinforcement Learning with Population-Coded Spiking Neural Network for Continuous Control","date":"2020-10-19","arxiv_id":"2010.09635","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-with-population#ran","syntology_url":"https://syntology.ai/paper/2010.09635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09635"}},"official":{"repos":["combra-lab/pop-spiking-deep-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/every-hidden-unit-maximizing-output-weights","slug":"every-hidden-unit-maximizing-output-weights","title":"Learning by Competition of Self-Interested Reinforcement Learning Agents","date":"2020-10-19","arxiv_id":"2010.09770","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-guided-open-attribute-value","slug":"knowledge-guided-open-attribute-value","title":"Knowledge-guided Open Attribute Value Extraction with Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09189","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/knowledge-guided-open-attribute-value#ran","syntology_url":"https://syntology.ai/paper/2010.09189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09189"}},"official":{"repos":["yeliu0930/Knowledge-guided-Open-Attribute-Value-Extraction-with-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-policy-optimization-with","slug":"model-based-policy-optimization-with","title":"Model-based Policy Optimization with Unsupervised Model Adaptation","date":"2020-10-19","arxiv_id":"2010.09546","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/model-based-policy-optimization-with#ran","syntology_url":"https://syntology.ai/paper/2010.09546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09546"}},"official":{"repos":["RockySJ/ampo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/approximate-information-state-for-approximate","slug":"approximate-information-state-for-approximate","title":"Approximate information state for approximate planning and reinforcement learning in partially observed systems","date":"2020-10-17","arxiv_id":"2010.08843","repositories_listed":1,"syntology":null},{"url":"/paper/a-game-theoretic-analysis-of-networked-system","slug":"a-game-theoretic-analysis-of-networked-system","title":"A game-theoretic analysis of networked system control for common-pool resource management using multi-agent reinforcement learning","date":"2020-10-15","arxiv_id":"2010.07777","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-game-theoretic-analysis-of-networked-system#ran","syntology_url":"https://syntology.ai/paper/2010.07777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07777"}},"official":{"repos":["instadeepai/EGTA-NMARL"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-alternative-to-backpropagation-in-deep","slug":"an-alternative-to-backpropagation-in-deep","title":"MAP Propagation Algorithm: Faster Learning with a Team of Reinforcement Learning Agents","date":"2020-10-15","arxiv_id":"2010.07893","repositories_listed":1,"syntology":null},{"url":"/paper/human-guided-robot-behavior-learning-a-gan","slug":"human-guided-robot-behavior-learning-a-gan","title":"Human-guided Robot Behavior Learning: A GAN-assisted Preference-based Reinforcement Learning Approach","date":"2020-10-15","arxiv_id":"2010.07467","repositories_listed":1,"syntology":null},{"url":"/paper/masked-contrastive-representation-learning","slug":"masked-contrastive-representation-learning","title":"Masked Contrastive Representation Learning for Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.07470","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/masked-contrastive-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2010.07470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07470"}},"official":{"repos":["teslacool/m-curl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-trust-region-policy-optimization","slug":"multi-agent-trust-region-policy-optimization","title":"Multi-Agent Trust Region Policy Optimization","date":"2020-10-15","arxiv_id":"2010.07916","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":7,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-trust-region-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2010.07916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07916"}},"official":null}},{"url":"/paper/multi-task-deep-reinforcement-learning-with-1","slug":"multi-task-deep-reinforcement-learning-with-1","title":"Knowledge Transfer in Multi-Task Deep Reinforcement Learning for Continuous Control","date":"2020-10-15","arxiv_id":"2010.07494","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2010.07494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07494"}},"official":null}},{"url":"/paper/safe-model-based-reinforcement-learning-with-1","slug":"safe-model-based-reinforcement-learning-with-1","title":"Constrained Model-based Reinforcement Learning with Robust Cross-Entropy Method","date":"2020-10-15","arxiv_id":"2010.07968","repositories_listed":1,"syntology":null},{"url":"/paper/modeling-protagonist-emotions-for-emotion","slug":"modeling-protagonist-emotions-for-emotion","title":"Modeling Protagonist Emotions for Emotion-Aware Storytelling","date":"2020-10-14","arxiv_id":"2010.06822","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-temporal-logic","slug":"reinforcement-learning-based-temporal-logic","title":"Reinforcement Learning Based Temporal Logic Control with Maximum Probabilistic Satisfaction","date":"2020-10-14","arxiv_id":"2010.06797","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-real-time","slug":"deep-reinforcement-learning-for-real-time","title":"Deep Reinforcement Learning for Real-Time Optimization of Pumps in Water Distribution Systems","date":"2020-10-13","arxiv_id":"2010.06460","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-wasserstein-natural-gradients-for-1","slug":"efficient-wasserstein-natural-gradients-for-1","title":"Efficient Wasserstein Natural Gradients for Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05380","repositories_listed":1,"syntology":null},{"url":"/paper/human-centric-dialog-training-via-offline","slug":"human-centric-dialog-training-via-offline","title":"Human-centric Dialog Training via Offline Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05848","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-dialog-training-via-offline#ran","syntology_url":"https://syntology.ai/paper/2010.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05848"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/contrastive-explanations-for-reinforcement-2","slug":"contrastive-explanations-for-reinforcement-2","title":"Contrastive Explanations for Reinforcement Learning via Embedded Self Predictions","date":"2020-10-11","arxiv_id":"2010.05180","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/contrastive-explanations-for-reinforcement-2#ran","syntology_url":"https://syntology.ai/paper/2010.05180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05180"}},"official":{"repos":["SuerpX/Embedded-Self-Predictions"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distributed-resource-allocation-with-multi","slug":"distributed-resource-allocation-with-multi","title":"Distributed Resource Allocation with Multi-Agent Deep Reinforcement Learning for 5G-V2V Communication","date":"2020-10-11","arxiv_id":"2010.05290","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolutional-value-decomposition-in-1","slug":"graph-convolutional-value-decomposition-in-1","title":"Graph Convolutional Value Decomposition in Multi-Agent Reinforcement Learning","date":"2020-10-09","arxiv_id":"2010.04740","repositories_listed":1,"syntology":null},{"url":"/paper/instance-weighted-incremental-evolution","slug":"instance-weighted-incremental-evolution","title":"Instance Weighted Incremental Evolution Strategies for Reinforcement Learning in Dynamic Environments","date":"2020-10-09","arxiv_id":"2010.04605","repositories_listed":1,"syntology":null},{"url":"/paper/land-learning-to-navigate-from-disengagements","slug":"land-learning-to-navigate-from-disengagements","title":"LaND: Learning to Navigate from Disengagements","date":"2020-10-09","arxiv_id":"2010.04689","repositories_listed":1,"syntology":null},{"url":"/paper/information-driven-adaptive-sensing-based-on","slug":"information-driven-adaptive-sensing-based-on","title":"Information-Driven Adaptive Sensing Based on Deep Reinforcement Learning","date":"2020-10-08","arxiv_id":"2010.04112","repositories_listed":1,"syntology":null},{"url":"/paper/maximum-reward-formulation-in-reinforcement-1","slug":"maximum-reward-formulation-in-reinforcement-1","title":"Maximum Reward Formulation In Reinforcement Learning","date":"2020-10-08","arxiv_id":"2010.03744","repositories_listed":1,"syntology":null},{"url":"/paper/diverse-exploration-via-infomax-options-1","slug":"diverse-exploration-via-infomax-options-1","title":"Learning Diverse Options via InfoMax Termination Critic","date":"2020-10-06","arxiv_id":"2010.02756","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diverse-exploration-via-infomax-options-1#ran","syntology_url":"https://syntology.ai/paper/2010.02756","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02756"}},"official":{"repos":["kngwyu/infomax-option-critic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-mask-generator-learning-to-generate","slug":"neural-mask-generator-learning-to-generate","title":"Neural Mask Generator: Learning to Generate Adaptive Word Maskings for Language Model Adaptation","date":"2020-10-06","arxiv_id":"2010.02705","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-for-1","slug":"a-reinforcement-learning-approach-for-1","title":"A Reinforcement Learning Approach for Rebalancing Electric Vehicle Sharing Systems","date":"2020-10-05","arxiv_id":"2010.02369","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-fiction-game-playing-as-multi","slug":"interactive-fiction-game-playing-as-multi","title":"Interactive Fiction Game Playing as Multi-Paragraph Reading Comprehension with Reinforcement Learning","date":"2020-10-05","arxiv_id":"2010.02386","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generalize-for-sequential","slug":"learning-to-generalize-for-sequential","title":"Learning to Generalize for Sequential Decision Making","date":"2020-10-05","arxiv_id":"2010.02229","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-of-compositional-task-1","slug":"meta-learning-of-compositional-task-1","title":"Meta-Learning of Structured Task Distributions in Humans and Machines","date":"2020-10-05","arxiv_id":"2010.02317","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-fully-offline-meta-reinforcement","slug":"efficient-fully-offline-meta-reinforcement","title":"FOCAL: Efficient Fully-Offline Meta-Reinforcement Learning via Distance Metric Learning and Behavior Regularization","date":"2020-10-02","arxiv_id":"2010.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-fully-offline-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01112"}},"official":{"repos":["FOCAL-ICLR/FOCAL-ICLR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploration-in-approximate-hyper-state-space","slug":"exploration-in-approximate-hyper-state-space","title":"Exploration in Approximate Hyper-State Space for Meta Reinforcement Learning","date":"2020-10-02","arxiv_id":"2010.01062","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/exploration-in-approximate-hyper-state-space#ran","syntology_url":"https://syntology.ai/paper/2010.01062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01062"}},"official":{"repos":["lmzintgraf/hyperx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/self-play-reinforcement-learning-for-fast","slug":"self-play-reinforcement-learning-for-fast","title":"Self-Play Reinforcement Learning for Fast Image Retargeting","date":"2020-10-02","arxiv_id":"2010.00909","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-efficient","slug":"deep-reinforcement-learning-for-efficient","title":"Deep Reinforcement Learning for Efficient Measurement of Quantum Devices","date":"2020-09-30","arxiv_id":"2009.14825","repositories_listed":1,"syntology":null},{"url":"/paper/learning-rewards-from-linguistic-feedback","slug":"learning-rewards-from-linguistic-feedback","title":"Learning Rewards from Linguistic Feedback","date":"2020-09-30","arxiv_id":"2009.14715","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-swim-in-potential-flow","slug":"learning-to-swim-in-potential-flow","title":"Learning to swim in potential flow","date":"2020-09-30","arxiv_id":"2009.14280","repositories_listed":1,"syntology":null},{"url":"/paper/multi-document-summarization-with-maximal","slug":"multi-document-summarization-with-maximal","title":"Multi-document Summarization with Maximal Marginal Relevance-guided Reinforcement Learning","date":"2020-09-30","arxiv_id":"2010.00117","repositories_listed":1,"syntology":null},{"url":"/paper/towards-effective-context-for-meta","slug":"towards-effective-context-for-meta","title":"Towards Effective Context for Meta-Reinforcement Learning: an Approach based on Contrastive Learning","date":"2020-09-29","arxiv_id":"2009.13891","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-effective-context-for-meta#ran","syntology_url":"https://syntology.ai/paper/2009.13891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.13891"}},"official":{"repos":["TJU-DRL-LAB/self-supervised-rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neurosymbolic-reinforcement-learning-with","slug":"neurosymbolic-reinforcement-learning-with","title":"Neurosymbolic Reinforcement Learning with Formally Verified Exploration","date":"2020-09-26","arxiv_id":"2009.12612","repositories_listed":1,"syntology":null},{"url":"/paper/an-automatic-cost-learning-framework-for","slug":"an-automatic-cost-learning-framework-for","title":"An Automatic Cost Learning Framework for Image Steganography Using Deep Reinforcement Learning","date":"2020-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continual-model-based-reinforcement-learning","slug":"continual-model-based-reinforcement-learning","title":"Continual Model-Based Reinforcement Learning with Hypernetworks","date":"2020-09-25","arxiv_id":"2009.11997","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-relational-deep-reinforcement","slug":"symbolic-relational-deep-reinforcement","title":"Symbolic Relational Deep Reinforcement Learning based on Graph Neural Networks and Autoregressive Policy Decomposition","date":"2020-09-25","arxiv_id":"2009.12462","repositories_listed":1,"syntology":null},{"url":"/paper/certrl-formalizing-convergence-proofs-for","slug":"certrl-formalizing-convergence-proofs-for","title":"CertRL: Formalizing Convergence Proofs for Value and Policy Iteration in Coq","date":"2020-09-23","arxiv_id":"2009.11403","repositories_listed":1,"syntology":null},{"url":"/paper/a-centralised-soft-actor-critic-deep","slug":"a-centralised-soft-actor-critic-deep","title":"A Centralised Soft Actor Critic Deep Reinforcement Learning Approach to District Demand Side Management through CityLearn","date":"2020-09-22","arxiv_id":"2009.10562","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-supervised-learning-and","slug":"rethinking-supervised-learning-and","title":"Rethinking Supervised Learning and Reinforcement Learning in Task-Oriented Dialogue Systems","date":"2020-09-21","arxiv_id":"2009.09781","repositories_listed":1,"syntology":null},{"url":"/paper/rl-star-platform-reinforcement-learning-for","slug":"rl-star-platform-reinforcement-learning-for","title":"RL STaR Platform: Reinforcement Learning for Simulation based Training of Robots","date":"2020-09-21","arxiv_id":"2009.09595","repositories_listed":1,"syntology":null},{"url":"/paper/structure-guided-processing-path-optimization","slug":"structure-guided-processing-path-optimization","title":"Deep Reinforcement Learning Methods for Structure-Guided Processing Path Optimization","date":"2020-09-21","arxiv_id":"2009.09706","repositories_listed":1,"syntology":null},{"url":"/paper/grac-self-guided-and-self-regularized-actor","slug":"grac-self-guided-and-self-regularized-actor","title":"GRAC: Self-Guided and Self-Regularized Actor-Critic","date":"2020-09-18","arxiv_id":"2009.08973","repositories_listed":1,"syntology":null}],"record_sha256":"613988120728ad4ff8a2649699b5956b70501107a09136b6b15096c9dec75fbe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}