{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/33","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":33,"pages_in_order":135,"rows_per_page":100,"rows":[3201,3300],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/32","next":"/task/reinforcement-learning-2/papers/34","papers":[{"url":"/paper/rlzoo-a-comprehensive-and-adaptive","slug":"rlzoo-a-comprehensive-and-adaptive","title":"Efficient Reinforcement Learning Development with RLzoo","date":"2020-09-18","arxiv_id":"2009.08644","repositories_listed":1,"syntology":null},{"url":"/paper/finding-effective-security-strategies-through","slug":"finding-effective-security-strategies-through","title":"Finding Effective Security Strategies through Reinforcement Learning and Self-Play","date":"2020-09-17","arxiv_id":"2009.08120","repositories_listed":1,"syntology":null},{"url":"/paper/meta-aad-active-anomaly-detection-with-deep","slug":"meta-aad-active-anomaly-detection-with-deep","title":"Meta-AAD: Active Anomaly Detection with Deep Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.07415","repositories_listed":1,"syntology":null},{"url":"/paper/deep-actor-critic-learning-for-distributed","slug":"deep-actor-critic-learning-for-distributed","title":"Deep Actor-Critic Learning for Distributed Power Control in Wireless Mobile Networks","date":"2020-09-14","arxiv_id":"2009.06681","repositories_listed":1,"syntology":null},{"url":"/paper/vacsim-learning-effective-strategies-for","slug":"vacsim-learning-effective-strategies-for","title":"VacSIM: Learning Effective Strategies for COVID-19 Vaccine Distribution using Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.06602","repositories_listed":1,"syntology":null},{"url":"/paper/physically-embedded-planning-problems-new","slug":"physically-embedded-planning-problems-new","title":"Physically Embedded Planning Problems: New Challenges for Reinforcement Learning","date":"2020-09-11","arxiv_id":"2009.05524","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-optimal-frequency","slug":"reinforcement-learning-for-optimal-frequency","title":"Reinforcement Learning for Optimal Primary Frequency Control: A Lyapunov Approach","date":"2020-09-11","arxiv_id":"2009.05654","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-preserving-reinforcement-learning","slug":"semantic-preserving-reinforcement-learning","title":"Semantic-preserving Reinforcement Learning Attack Against Graph Neural Networks for Malware Detection","date":"2020-09-11","arxiv_id":"2009.05602","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-reinforcement-learning-with","slug":"a-framework-for-reinforcement-learning-with","title":"A framework for reinforcement learning with autocorrelated actions","date":"2020-09-10","arxiv_id":"2009.04777","repositories_listed":1,"syntology":null},{"url":"/paper/tripletree-a-versatile-interpretable","slug":"tripletree-a-versatile-interpretable","title":"TripleTree: A Versatile Interpretable Representation of Black Box Agents and their Environments","date":"2020-09-10","arxiv_id":"2009.04743","repositories_listed":1,"syntology":null},{"url":"/paper/dynode-neural-ordinary-differential-equations","slug":"dynode-neural-ordinary-differential-equations","title":"DyNODE: Neural Ordinary Differential Equations for Dynamics Modeling in Continuous Control","date":"2020-09-09","arxiv_id":"2009.04278","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynode-neural-ordinary-differential-equations#ran","syntology_url":"https://syntology.ai/paper/2009.04278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.04278"}},"official":{"repos":["vmartinezalvarez/DyNODE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-inverse-reinforcement-learning-for","slug":"bayesian-inverse-reinforcement-learning-for","title":"Bayesian Inverse Reinforcement Learning for Collective Animal Movement","date":"2020-09-08","arxiv_id":"2009.04003","repositories_listed":1,"syntology":null},{"url":"/paper/deep-active-inference-for-partially","slug":"deep-active-inference-for-partially","title":"Deep Active Inference for Partially Observable MDPs","date":"2020-09-08","arxiv_id":"2009.03622","repositories_listed":1,"syntology":null},{"url":"/paper/confuciux-autonomous-hardware-resource","slug":"confuciux-autonomous-hardware-resource","title":"ConfuciuX: Autonomous Hardware Resource Assignment for DNN Accelerators using Reinforcement Learning","date":"2020-09-04","arxiv_id":"2009.02010","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/confuciux-autonomous-hardware-resource#ran","syntology_url":"https://syntology.ai/paper/2009.02010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.02010"}},"official":{"repos":["maestro-project/confuciux"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/drle-decentralized-reinforcement-learning-at","slug":"drle-decentralized-reinforcement-learning-at","title":"DRLE: Decentralized Reinforcement Learning at the Edge for Traffic Light Control in the IoV","date":"2020-09-03","arxiv_id":"2009.01502","repositories_listed":1,"syntology":null},{"url":"/paper/optimality-based-analysis-of-xcsf-compaction","slug":"optimality-based-analysis-of-xcsf-compaction","title":"Optimality-based Analysis of XCSF Compaction in Discrete Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01476","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-automated-deep-reinforcement","slug":"sample-efficient-automated-deep-reinforcement","title":"Sample-Efficient Automated Deep Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01555","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-automated-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.01555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.01555"}},"official":{"repos":["automl/SEARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-model-based-stochastic-value-gradient","slug":"on-the-model-based-stochastic-value-gradient","title":"On the model-based stochastic value gradient for continuous reinforcement learning","date":"2020-08-28","arxiv_id":"2008.12775","repositories_listed":1,"syntology":null},{"url":"/paper/autofs-automated-feature-selection-via","slug":"autofs-automated-feature-selection-via","title":"AutoFS: Automated Feature Selection via Diversity-aware Interactive Reinforcement Learning","date":"2020-08-27","arxiv_id":"2008.12001","repositories_listed":1,"syntology":null},{"url":"/paper/query-focused-multi-document-summarisation-of","slug":"query-focused-multi-document-summarisation-of","title":"Query Focused Multi-document Summarisation of Biomedical Texts","date":"2020-08-27","arxiv_id":"2008.11986","repositories_listed":1,"syntology":null},{"url":"/paper/query-focused-multi-document-summarisation-of-1","slug":"query-focused-multi-document-summarisation-of-1","title":"Query Focused Multi-document Summarisation of Biomedical Texts: Macquarie Universiy and the Australian National University at BioASQ8b","date":"2020-08-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-off-policy-with-online-planning","slug":"learning-off-policy-with-online-planning","title":"Learning Off-Policy with Online Planning","date":"2020-08-23","arxiv_id":"2008.10066","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-off-policy-with-online-planning#ran","syntology_url":"https://syntology.ai/paper/2008.10066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.10066"}},"official":null}},{"url":"/paper/a-composable-specification-language-for-1","slug":"a-composable-specification-language-for-1","title":"A Composable Specification Language for Reinforcement Learning Tasks","date":"2020-08-21","arxiv_id":"2008.09293","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-composable-specification-language-for-1#ran","syntology_url":"https://syntology.ai/paper/2008.09293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09293"}},"official":{"repos":["keyshor/spectrl_tool"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/social-aware-incentive-mechanism-for","slug":"social-aware-incentive-mechanism-for","title":"Social-Aware Incentive Mechanism for VehicularCrowdsensing by Deep Reinforcement Learning","date":"2020-08-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/optimization-of-operation-parameters-towards","slug":"optimization-of-operation-parameters-towards","title":"Optimal control towards sustainable wastewater treatment plants based on multi-agent reinforcement learning","date":"2020-08-19","arxiv_id":"2008.10417","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-low-thrust","slug":"reinforcement-learning-for-low-thrust","title":"Reinforcement Learning for Low-Thrust Trajectory Design of Interplanetary Missions","date":"2020-08-19","arxiv_id":"2008.08501","repositories_listed":1,"syntology":null},{"url":"/paper/communicative-reinforcement-learning-agents","slug":"communicative-reinforcement-learning-agents","title":"Communicative Reinforcement Learning Agents for Landmark Detection in Brain Images","date":"2020-08-18","arxiv_id":"2008.08055","repositories_listed":1,"syntology":null},{"url":"/paper/learning-fair-policies-in-multiobjective-deep","slug":"learning-fair-policies-in-multiobjective-deep","title":"Learning Fair Policies in Multiobjective (Deep) Reinforcement Learning with Average and Discounted Rewards","date":"2020-08-18","arxiv_id":"2008.07773","repositories_listed":1,"syntology":null},{"url":"/paper/towards-closing-the-sim-to-real-gap-in","slug":"towards-closing-the-sim-to-real-gap-in","title":"Towards Closing the Sim-to-Real Gap in Collaborative Multi-Robot Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07875","repositories_listed":1,"syntology":null},{"url":"/paper/supersuit-simple-microwrappers-for","slug":"supersuit-simple-microwrappers-for","title":"SuperSuit: Simple Microwrappers for Reinforcement Learning Environments","date":"2020-08-17","arxiv_id":"2008.08932","repositories_listed":1,"syntology":null},{"url":"/paper/cautious-adaptation-for-reinforcement","slug":"cautious-adaptation-for-reinforcement","title":"Cautious Adaptation For Reinforcement Learning in Safety-Critical Settings","date":"2020-08-15","arxiv_id":"2008.06622","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-in-constrained","slug":"safe-reinforcement-learning-in-constrained","title":"Safe Reinforcement Learning in Constrained Markov Decision Processes","date":"2020-08-15","arxiv_id":"2008.06626","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-reinforcement-learning-in-constrained#ran","syntology_url":"https://syntology.ai/paper/2008.06626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06626"}},"official":{"repos":["akifumi-wachi-4/safe_near_optimal_mdp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-cross-entropy-method-for","slug":"sample-efficient-cross-entropy-method-for","title":"Sample-efficient Cross-Entropy Method for Real-time Planning","date":"2020-08-14","arxiv_id":"2008.06389","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sample-efficient-cross-entropy-method-for#ran","syntology_url":"https://syntology.ai/paper/2008.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06389"}},"official":{"repos":["martius-lab/iCEM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safepilco-a-software-tool-for-safe-and-data","slug":"safepilco-a-software-tool-for-safe-and-data","title":"SafePILCO: a software tool for safe and data-efficient policy synthesis","date":"2020-08-07","arxiv_id":"2008.03273","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-variational-model-based","slug":"contrastive-variational-model-based","title":"Contrastive Variational Reinforcement Learning for Complex Observations","date":"2020-08-06","arxiv_id":"2008.02430","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-tactile","slug":"deep-reinforcement-learning-for-tactile","title":"Deep Reinforcement Learning for Tactile Robotics: Learning to Type on a Braille Keyboard","date":"2020-08-06","arxiv_id":"2008.02646","repositories_listed":1,"syntology":null},{"url":"/paper/mixed-initiative-level-design-with-rl-brush","slug":"mixed-initiative-level-design-with-rl-brush","title":"Mixed-Initiative Level Design with RL Brush","date":"2020-08-06","arxiv_id":"2008.02778","repositories_listed":1,"syntology":null},{"url":"/paper/the-emergence-of-adversarial-communication-in","slug":"the-emergence-of-adversarial-communication-in","title":"The Emergence of Adversarial Communication in Multi-Agent Reinforcement Learning","date":"2020-08-06","arxiv_id":"2008.02616","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-emergence-of-adversarial-communication-in#ran","syntology_url":"https://syntology.ai/paper/2008.02616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02616"}},"official":{"repos":["proroklab/adversarial_comms"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-adaptive-task-offloading-in-edge","slug":"fast-adaptive-task-offloading-in-edge","title":"Fast Adaptive Task Offloading in Edge Computing based on Meta Reinforcement Learning","date":"2020-08-05","arxiv_id":"2008.02033","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-epidemic-control-saving-both-lives","slug":"reinforced-epidemic-control-saving-both-lives","title":"Reinforced Epidemic Control: Saving Both Lives and Economy","date":"2020-08-04","arxiv_id":"2008.01257","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-using","slug":"robust-reinforcement-learning-using","title":"Robust Reinforcement Learning using Adversarial Populations","date":"2020-08-04","arxiv_id":"2008.01825","repositories_listed":1,"syntology":null},{"url":"/paper/queueing-network-controls-via-deep","slug":"queueing-network-controls-via-deep","title":"Queueing Network Controls via Deep Reinforcement Learning","date":"2020-07-31","arxiv_id":"2008.01644","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/queueing-network-controls-via-deep#ran","syntology_url":"https://syntology.ai/paper/2008.01644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01644"}},"official":null}},{"url":"/paper/pixl2r-guiding-reinforcement-learning-using","slug":"pixl2r-guiding-reinforcement-learning-using","title":"PixL2R: Guiding Reinforcement Learning Using Natural Language by Mapping Pixels to Rewards","date":"2020-07-30","arxiv_id":"2007.15543","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pixl2r-guiding-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2007.15543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.15543"}},"official":{"repos":["prasoongoyal/PixL2R"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lifelong-incremental-reinforcement-learning","slug":"lifelong-incremental-reinforcement-learning","title":"Lifelong Incremental Reinforcement Learning with Online Bayesian Inference","date":"2020-07-28","arxiv_id":"2007.14196","repositories_listed":1,"syntology":null},{"url":"/paper/multi-step-reinforcement-learning-for-single","slug":"multi-step-reinforcement-learning-for-single","title":"Multi-Step Reinforcement Learning for Single Image Super-Resolution","date":"2020-07-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/combining-deep-reinforcement-learning-and-1","slug":"combining-deep-reinforcement-learning-and-1","title":"Combining Deep Reinforcement Learning and Search for Imperfect-Information Games","date":"2020-07-27","arxiv_id":"2007.13544","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/combining-deep-reinforcement-learning-and-1#ran","syntology_url":"https://syntology.ai/paper/2007.13544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.13544"}},"official":{"repos":["facebookresearch/rebel"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-discovery-of-local-rules-for","slug":"automated-discovery-of-local-rules-for","title":"Automated Discovery of Local Rules for Desired Collective-Level Behavior Through Reinforcement Learning","date":"2020-07-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/human-preference-scaling-with-demonstrations","slug":"human-preference-scaling-with-demonstrations","title":"Weak Human Preference Supervision For Deep Reinforcement Learning","date":"2020-07-25","arxiv_id":"2007.12904","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-exploration-under-uncertainty-via","slug":"autonomous-exploration-under-uncertainty-via","title":"Autonomous Exploration Under Uncertainty via Deep Reinforcement Learning on Graphs","date":"2020-07-24","arxiv_id":"2007.12640","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-robust-optimization-for-imitation","slug":"bayesian-robust-optimization-for-imitation","title":"Bayesian Robust Optimization for Imitation Learning","date":"2020-07-24","arxiv_id":"2007.12315","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayesian-robust-optimization-for-imitation#ran","syntology_url":"https://syntology.ai/paper/2007.12315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12315"}},"official":{"repos":["dsbrown1331/broil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-inverse-reinforcement-learning-for","slug":"deep-inverse-reinforcement-learning-for","title":"Deep Inverse Reinforcement Learning for Structural Evolution of Small Molecules","date":"2020-07-24","arxiv_id":"2008.11804","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-with-3","slug":"distributional-reinforcement-learning-with-3","title":"Distributional Reinforcement Learning via Moment Matching","date":"2020-07-24","arxiv_id":"2007.12354","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distributional-reinforcement-learning-with-3#ran","syntology_url":"https://syntology.ai/paper/2007.12354","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12354"}},"official":{"repos":["thanhnguyentang/mmdrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/challenging-common-bolus-advisor-for-self","slug":"challenging-common-bolus-advisor-for-self","title":"Challenging common bolus advisor for self-monitoring type-I diabetes patients using Reinforcement Learning","date":"2020-07-23","arxiv_id":"2007.11880","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-traffic-control-with-deep","slug":"adaptive-traffic-control-with-deep","title":"Adaptive Traffic Control with Deep Reinforcement Learning:Towards State-of-the-art and Beyond","date":"2020-07-21","arxiv_id":"2007.10960","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-deep-reinforcement-learning","slug":"integrating-deep-reinforcement-learning","title":"Integrating Deep Reinforcement Learning Networks with Health System Simulations","date":"2020-07-21","arxiv_id":"2008.07434","repositories_listed":1,"syntology":null},{"url":"/paper/battlesnake-challenge-a-multi-agent","slug":"battlesnake-challenge-a-multi-agent","title":"Battlesnake Challenge: A Multi-agent Reinforcement Learning Playground with Human-in-the-loop","date":"2020-07-20","arxiv_id":"2007.10504","repositories_listed":1,"syntology":null},{"url":"/paper/structure-mapping-for-transferability-of","slug":"structure-mapping-for-transferability-of","title":"Structure Mapping for Transferability of Causal Models","date":"2020-07-18","arxiv_id":"2007.09445","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/structure-mapping-for-transferability-of#ran","syntology_url":"https://syntology.ai/paper/2007.09445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.09445"}},"official":{"repos":["Information-Fusion-Lab-Umass/causal_transfer_learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/off-policy-reinforcement-learning-for","slug":"off-policy-reinforcement-learning-for","title":"Off-Policy Reinforcement Learning for Efficient and Effective GAN Architecture Search","date":"2020-07-17","arxiv_id":"2007.09180","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/off-policy-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2007.09180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.09180"}},"official":{"repos":["Yuantian013/E2GAN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/collision-avoidance-robotics-via-meta","slug":"collision-avoidance-robotics-via-meta","title":"Collision Avoidance Robotics Via Meta-Learning (CARML)","date":"2020-07-16","arxiv_id":"2007.08616","repositories_listed":1,"syntology":null},{"url":"/paper/provably-good-batch-reinforcement-learning","slug":"provably-good-batch-reinforcement-learning","title":"Provably Good Batch Reinforcement Learning Without Great Exploration","date":"2020-07-16","arxiv_id":"2007.08202","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provably-good-batch-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.08202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.08202"}},"official":null}},{"url":"/paper/weighing-counts-sequential-crowd-counting-by","slug":"weighing-counts-sequential-crowd-counting-by","title":"Weighing Counts: Sequential Crowd Counting by Reinforcement Learning","date":"2020-07-16","arxiv_id":"2007.08260","repositories_listed":1,"syntology":null},{"url":"/paper/developmental-reinforcement-learning-of","slug":"developmental-reinforcement-learning-of","title":"Developmental Reinforcement Learning of Control Policy of a Quadcopter UAV with Thrust Vectoring Rotors","date":"2020-07-15","arxiv_id":"2007.07793","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-reward-functions-using-anchor","slug":"identifying-reward-functions-using-anchor","title":"Deep PQR: Solving Inverse Reinforcement Learning using Anchor Actions","date":"2020-07-15","arxiv_id":"2007.07443","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-of-musculoskeletal","slug":"reinforcement-learning-of-musculoskeletal","title":"Reinforcement Learning of Musculoskeletal Control from Functional Simulations","date":"2020-07-13","arxiv_id":"2007.06669","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-reinforcement-learning-with-1","slug":"data-efficient-reinforcement-learning-with-1","title":"Data-Efficient Reinforcement Learning with Self-Predictive Representations","date":"2020-07-12","arxiv_id":"2007.05929","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/data-efficient-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2007.05929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05929"}},"official":{"repos":["mila-iqia/spr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/xcs-as-a-reinforcement-learning-approach-to","slug":"xcs-as-a-reinforcement-learning-approach-to","title":"XCS as a reinforcement learning approach to automatic test case prioritization","date":"2020-07-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/long-term-planning-with-deep-reinforcement","slug":"long-term-planning-with-deep-reinforcement","title":"Long-Term Planning with Deep Reinforcement Learning on Autonomous Drones","date":"2020-07-11","arxiv_id":"2007.05694","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/long-term-planning-with-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.05694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05694"}},"official":{"repos":["ugurkanates/NeurIRS2019DroneChallengeRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-trained-word-embeddings-for-goal","slug":"pre-trained-word-embeddings-for-goal","title":"Pre-trained Word Embeddings for Goal-conditional Transfer Learning in Reinforcement Learning","date":"2020-07-10","arxiv_id":"2007.05196","repositories_listed":1,"syntology":null},{"url":"/paper/fast-reinforcement-learning-with-generalized","slug":"fast-reinforcement-learning-with-generalized","title":"Fast reinforcement learning with generalized policy updates","date":"2020-07-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-retrospective-knowledge-with-reverse","slug":"learning-retrospective-knowledge-with-reverse","title":"Learning Retrospective Knowledge with Reverse Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.06703","repositories_listed":1,"syntology":null},{"url":"/paper/sunrise-a-simple-unified-framework-for","slug":"sunrise-a-simple-unified-framework-for","title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04938","repositories_listed":1,"syntology":null},{"url":"/paper/provably-safe-pac-mdp-exploration-using","slug":"provably-safe-pac-mdp-exploration-using","title":"Provably Safe PAC-MDP Exploration Using Analogies","date":"2020-07-07","arxiv_id":"2007.03574","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-data-augmentation-using","slug":"counterfactual-data-augmentation-using","title":"Counterfactual Data Augmentation using Locally Factored Dynamics","date":"2020-07-06","arxiv_id":"2007.02863","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-data-augmentation-using#ran","syntology_url":"https://syntology.ai/paper/2007.02863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02863"}},"official":null}},{"url":"/paper/enhancing-sat-solvers-with-glue-variable","slug":"enhancing-sat-solvers-with-glue-variable","title":"Enhancing SAT solvers with glue variable predictions","date":"2020-07-06","arxiv_id":"2007.02559","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":2,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-sat-solvers-with-glue-variable#ran","syntology_url":"https://syntology.ai/paper/2007.02559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02559"}},"official":{"repos":["jesse-michael-han/neuro-cadical"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-implicit-credit-assignment-for-multi","slug":"learning-implicit-credit-assignment-for-multi","title":"Learning Implicit Credit Assignment for Cooperative Multi-Agent Reinforcement Learning","date":"2020-07-06","arxiv_id":"2007.02529","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-implicit-credit-assignment-for-multi#ran","syntology_url":"https://syntology.ai/paper/2007.02529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02529"}},"official":{"repos":["mzho7212/LICA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/lfq-online-learning-of-per-flow-queuing","slug":"lfq-online-learning-of-per-flow-queuing","title":"LFQ: Online Learning of Per-flow Queuing Policies using Deep Reinforcement Learning","date":"2020-07-06","arxiv_id":"2007.02735","repositories_listed":1,"syntology":null},{"url":"/paper/nappo-modular-and-scalable-reinforcement","slug":"nappo-modular-and-scalable-reinforcement","title":"Integrating Distributed Architectures in Highly Modular RL Libraries","date":"2020-07-06","arxiv_id":"2007.02622","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/nappo-modular-and-scalable-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.02622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02622"}},"official":null}},{"url":"/paper/discount-factor-as-a-regularizer-in","slug":"discount-factor-as-a-regularizer-in","title":"Discount Factor as a Regularizer in Reinforcement Learning","date":"2020-07-04","arxiv_id":"2007.02040","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discount-factor-as-a-regularizer-in#ran","syntology_url":"https://syntology.ai/paper/2007.02040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02040"}},"official":{"repos":["ron-amit/Discount_as_Regularizer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-inverse-reinforcement-learning-under","slug":"robust-inverse-reinforcement-learning-under","title":"Robust Inverse Reinforcement Learning under Transition Dynamics Mismatch","date":"2020-07-02","arxiv_id":"2007.01174","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/robust-inverse-reinforcement-learning-under#ran","syntology_url":"https://syntology.ai/paper/2007.01174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01174"}},"official":{"repos":["lviano/robustmce_irl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/verifiably-safe-exploration-for-end-to-end","slug":"verifiably-safe-exploration-for-end-to-end","title":"Verifiably Safe Exploration for End-to-End Reinforcement Learning","date":"2020-07-02","arxiv_id":"2007.01223","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-discretization-for-model-based","slug":"adaptive-discretization-for-model-based","title":"Adaptive Discretization for Model-Based Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.00717","repositories_listed":1,"syntology":null},{"url":"/paper/group-equivariant-deep-reinforcement-learning","slug":"group-equivariant-deep-reinforcement-learning","title":"Group Equivariant Deep Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.03437","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/group-equivariant-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.03437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03437"}},"official":{"repos":["arnab39/EquivariantDQN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reinforcement-learning-based-control-of","slug":"reinforcement-learning-based-control-of","title":"Reinforcement Learning based Control of Imitative Policies for Near-Accident Driving","date":"2020-07-01","arxiv_id":"2007.00178","repositories_listed":1,"syntology":null},{"url":"/paper/enforcing-almost-sure-reachability-in-pomdps","slug":"enforcing-almost-sure-reachability-in-pomdps","title":"Enforcing Almost-Sure Reachability in POMDPs","date":"2020-06-30","arxiv_id":"2007.00085","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-performance-of-reinforcement","slug":"evaluating-the-performance-of-reinforcement","title":"Evaluating the Performance of Reinforcement Learning Algorithms","date":"2020-06-30","arxiv_id":"2006.16958","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-for-semi","slug":"model-based-reinforcement-learning-for-semi","title":"Model-based Reinforcement Learning for Semi-Markov Decision Processes with Neural ODEs","date":"2020-06-29","arxiv_id":"2006.16210","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-semi#ran","syntology_url":"https://syntology.ai/paper/2006.16210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16210"}},"official":{"repos":["dtak/mbrl-smdp-ode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/image-classification-by-reinforcement","slug":"image-classification-by-reinforcement","title":"Image Classification by Reinforcement Learning with Two-State Q-Learning","date":"2020-06-28","arxiv_id":"2007.01298","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforced-model-for-zero-shot-cross-1","slug":"a-deep-reinforced-model-for-zero-shot-cross-1","title":"A Deep Reinforced Model for Zero-Shot Cross-Lingual Summarization with Bilingual Semantic Similarity Rewards","date":"2020-06-27","arxiv_id":"2006.15454","repositories_listed":1,"syntology":null},{"url":"/paper/explanation-augmented-feedback-in-human-in","slug":"explanation-augmented-feedback-in-human-in","title":"Widening the Pipeline in Human-Guided Reinforcement Learning with Explanation and Context-Aware Data Augmentation","date":"2020-06-26","arxiv_id":"2006.14804","repositories_listed":1,"syntology":null},{"url":"/paper/online-3d-bin-packing-with-constrained-deep","slug":"online-3d-bin-packing-with-constrained-deep","title":"Online 3D Bin Packing with Constrained Deep Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.14978","repositories_listed":1,"syntology":null},{"url":"/paper/what-can-i-do-here-a-theory-of-affordances-in","slug":"what-can-i-do-here-a-theory-of-affordances-in","title":"What can I do here? A Theory of Affordances in Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.15085","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-data-augmentation-for","slug":"automatic-data-augmentation-for","title":"Automatic Data Augmentation for Generalization in Deep Reinforcement Learning","date":"2020-06-23","arxiv_id":"2006.12862","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automatic-data-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2006.12862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.12862"}},"official":{"repos":["rraileanu/auto-drac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/experience-replay-with-likelihood-free","slug":"experience-replay-with-likelihood-free","title":"Experience Replay with Likelihood-free Importance Weights","date":"2020-06-23","arxiv_id":"2006.13169","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/experience-replay-with-likelihood-free#ran","syntology_url":"https://syntology.ai/paper/2006.13169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13169"}},"official":null}},{"url":"/paper/dm-control-software-and-tasks-for-continuous","slug":"dm-control-software-and-tasks-for-continuous","title":"dm_control: Software and Tasks for Continuous Control","date":"2020-06-22","arxiv_id":"2006.12983","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-via-curriculum","slug":"safe-reinforcement-learning-via-curriculum","title":"Safe Reinforcement Learning via Curriculum Induction","date":"2020-06-22","arxiv_id":"2006.12136","repositories_listed":1,"syntology":null},{"url":"/paper/automated-optical-multi-layer-design-via-deep","slug":"automated-optical-multi-layer-design-via-deep","title":"Automated Optical Multi-layer Design via Deep Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11940","repositories_listed":1,"syntology":null},{"url":"/paper/generating-adjacency-constrained-subgoals-in","slug":"generating-adjacency-constrained-subgoals-in","title":"Generating Adjacency-Constrained Subgoals in Hierarchical Reinforcement Learning","date":"2020-06-20","arxiv_id":"2006.11485","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generating-adjacency-constrained-subgoals-in#ran","syntology_url":"https://syntology.ai/paper/2006.11485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11485"}},"official":{"repos":["trzhang0116/HRAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/deep-implicit-coordination-graphs-for-multi","slug":"deep-implicit-coordination-graphs-for-multi","title":"Deep Implicit Coordination Graphs for Multi-agent Reinforcement Learning","date":"2020-06-19","arxiv_id":"2006.11438","repositories_listed":1,"syntology":null},{"url":"/paper/task-agnostic-online-reinforcement-learning","slug":"task-agnostic-online-reinforcement-learning","title":"Task-Agnostic Online Reinforcement Learning with an Infinite Mixture of Gaussian Processes","date":"2020-06-19","arxiv_id":"2006.11441","repositories_listed":1,"syntology":null},{"url":"/paper/dream-deep-regret-minimization-with-advantage","slug":"dream-deep-regret-minimization-with-advantage","title":"DREAM: Deep Regret minimization with Advantage baselines and Model-free learning","date":"2020-06-18","arxiv_id":"2006.10410","repositories_listed":1,"syntology":null}],"record_sha256":"3aa49ba5a78183b8f451ba09365aaeb78b70edd4560833d414d2b10330dd9bbf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}