{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/37","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":37,"pages_in_order":152,"rows_per_page":100,"rows":[3601,3700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/36","next":"/task/reinforcement-learning-1/papers/38","papers":[{"url":"/paper/meta-learning-of-compositional-task-1","slug":"meta-learning-of-compositional-task-1","title":"Meta-Learning of Structured Task Distributions in Humans and Machines","date":"2020-10-05","arxiv_id":"2010.02317","repositories_listed":1,"syntology":null},{"url":"/paper/policy-learning-using-weak-supervision-1","slug":"policy-learning-using-weak-supervision-1","title":"Policy Learning Using Weak Supervision","date":"2020-10-05","arxiv_id":"2010.01748","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-learning-using-weak-supervision-1#ran","syntology_url":"https://syntology.ai/paper/2010.01748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01748"}},"official":{"repos":["wangjksjtu/peerpl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-fully-offline-meta-reinforcement","slug":"efficient-fully-offline-meta-reinforcement","title":"FOCAL: Efficient Fully-Offline Meta-Reinforcement Learning via Distance Metric Learning and Behavior Regularization","date":"2020-10-02","arxiv_id":"2010.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-fully-offline-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01112"}},"official":{"repos":["FOCAL-ICLR/FOCAL-ICLR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploration-in-approximate-hyper-state-space","slug":"exploration-in-approximate-hyper-state-space","title":"Exploration in Approximate Hyper-State Space for Meta Reinforcement Learning","date":"2020-10-02","arxiv_id":"2010.01062","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/exploration-in-approximate-hyper-state-space#ran","syntology_url":"https://syntology.ai/paper/2010.01062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01062"}},"official":{"repos":["lmzintgraf/hyperx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-auxiliary-actor-critic-for-6d-robotic","slug":"goal-auxiliary-actor-critic-for-6d-robotic","title":"Goal-Auxiliary Actor-Critic for 6D Robotic Grasping with Point Clouds","date":"2020-10-02","arxiv_id":"2010.00824","repositories_listed":1,"syntology":null},{"url":"/paper/self-play-reinforcement-learning-for-fast","slug":"self-play-reinforcement-learning-for-fast","title":"Self-Play Reinforcement Learning for Fast Image Retargeting","date":"2020-10-02","arxiv_id":"2010.00909","repositories_listed":1,"syntology":null},{"url":"/paper/student-initiated-action-advising-via-advice","slug":"student-initiated-action-advising-via-advice","title":"Student-Initiated Action Advising via Advice Novelty","date":"2020-10-01","arxiv_id":"2010.00381","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-efficient","slug":"deep-reinforcement-learning-for-efficient","title":"Deep Reinforcement Learning for Efficient Measurement of Quantum Devices","date":"2020-09-30","arxiv_id":"2009.14825","repositories_listed":1,"syntology":null},{"url":"/paper/learning-rewards-from-linguistic-feedback","slug":"learning-rewards-from-linguistic-feedback","title":"Learning Rewards from Linguistic Feedback","date":"2020-09-30","arxiv_id":"2009.14715","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-swim-in-potential-flow","slug":"learning-to-swim-in-potential-flow","title":"Learning to swim in potential flow","date":"2020-09-30","arxiv_id":"2009.14280","repositories_listed":1,"syntology":null},{"url":"/paper/multi-document-summarization-with-maximal","slug":"multi-document-summarization-with-maximal","title":"Multi-document Summarization with Maximal Marginal Relevance-guided Reinforcement Learning","date":"2020-09-30","arxiv_id":"2010.00117","repositories_listed":1,"syntology":null},{"url":"/paper/a-traffic-light-dynamic-control-algorithm","slug":"a-traffic-light-dynamic-control-algorithm","title":"A Traffic Light Dynamic Control Algorithm with Deep Reinforcement Learning Based on GNN Prediction","date":"2020-09-29","arxiv_id":"2009.14627","repositories_listed":1,"syntology":null},{"url":"/paper/lucid-dreaming-for-experience-replay","slug":"lucid-dreaming-for-experience-replay","title":"Lucid Dreaming for Experience Replay: Refreshing Past States with the Current Policy","date":"2020-09-29","arxiv_id":"2009.13736","repositories_listed":1,"syntology":null},{"url":"/paper/towards-effective-context-for-meta","slug":"towards-effective-context-for-meta","title":"Towards Effective Context for Meta-Reinforcement Learning: an Approach based on Contrastive Learning","date":"2020-09-29","arxiv_id":"2009.13891","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-effective-context-for-meta#ran","syntology_url":"https://syntology.ai/paper/2009.13891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.13891"}},"official":{"repos":["TJU-DRL-LAB/self-supervised-rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neurosymbolic-reinforcement-learning-with","slug":"neurosymbolic-reinforcement-learning-with","title":"Neurosymbolic Reinforcement Learning with Formally Verified Exploration","date":"2020-09-26","arxiv_id":"2009.12612","repositories_listed":1,"syntology":null},{"url":"/paper/continual-model-based-reinforcement-learning","slug":"continual-model-based-reinforcement-learning","title":"Continual Model-Based Reinforcement Learning with Hypernetworks","date":"2020-09-25","arxiv_id":"2009.11997","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-relational-deep-reinforcement","slug":"symbolic-relational-deep-reinforcement","title":"Symbolic Relational Deep Reinforcement Learning based on Graph Neural Networks and Autoregressive Policy Decomposition","date":"2020-09-25","arxiv_id":"2009.12462","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrapped-q-learning-with-context-relevant","slug":"bootstrapped-q-learning-with-context-relevant","title":"Bootstrapped Q-learning with Context Relevant Observation Pruning to Generalize in Text-based Games","date":"2020-09-24","arxiv_id":"2009.11896","repositories_listed":1,"syntology":null},{"url":"/paper/certrl-formalizing-convergence-proofs-for","slug":"certrl-formalizing-convergence-proofs-for","title":"CertRL: Formalizing Convergence Proofs for Value and Policy Iteration in Coq","date":"2020-09-23","arxiv_id":"2009.11403","repositories_listed":1,"syntology":null},{"url":"/paper/a-centralised-soft-actor-critic-deep","slug":"a-centralised-soft-actor-critic-deep","title":"A Centralised Soft Actor Critic Deep Reinforcement Learning Approach to District Demand Side Management through CityLearn","date":"2020-09-22","arxiv_id":"2009.10562","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-supervised-learning-and","slug":"rethinking-supervised-learning-and","title":"Rethinking Supervised Learning and Reinforcement Learning in Task-Oriented Dialogue Systems","date":"2020-09-21","arxiv_id":"2009.09781","repositories_listed":1,"syntology":null},{"url":"/paper/rl-star-platform-reinforcement-learning-for","slug":"rl-star-platform-reinforcement-learning-for","title":"RL STaR Platform: Reinforcement Learning for Simulation based Training of Robots","date":"2020-09-21","arxiv_id":"2009.09595","repositories_listed":1,"syntology":null},{"url":"/paper/structure-guided-processing-path-optimization","slug":"structure-guided-processing-path-optimization","title":"Deep Reinforcement Learning Methods for Structure-Guided Processing Path Optimization","date":"2020-09-21","arxiv_id":"2009.09706","repositories_listed":1,"syntology":null},{"url":"/paper/grac-self-guided-and-self-regularized-actor","slug":"grac-self-guided-and-self-regularized-actor","title":"GRAC: Self-Guided and Self-Regularized Actor-Critic","date":"2020-09-18","arxiv_id":"2009.08973","repositories_listed":1,"syntology":null},{"url":"/paper/rlzoo-a-comprehensive-and-adaptive","slug":"rlzoo-a-comprehensive-and-adaptive","title":"Efficient Reinforcement Learning Development with RLzoo","date":"2020-09-18","arxiv_id":"2009.08644","repositories_listed":1,"syntology":null},{"url":"/paper/competitiveness-of-map-elites-against","slug":"competitiveness-of-map-elites-against","title":"Competitiveness of MAP-Elites against Proximal Policy Optimization on locomotion tasks in deterministic simulations","date":"2020-09-17","arxiv_id":"2009.08438","repositories_listed":1,"syntology":null},{"url":"/paper/finding-effective-security-strategies-through","slug":"finding-effective-security-strategies-through","title":"Finding Effective Security Strategies through Reinforcement Learning and Self-Play","date":"2020-09-17","arxiv_id":"2009.08120","repositories_listed":1,"syntology":null},{"url":"/paper/srec-proactive-self-remedy-of-energy","slug":"srec-proactive-self-remedy-of-energy","title":"SREC: Proactive Self-Remedy of Energy-Constrained UAV-Based Networks via Deep Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08528","repositories_listed":1,"syntology":null},{"url":"/paper/meta-aad-active-anomaly-detection-with-deep","slug":"meta-aad-active-anomaly-detection-with-deep","title":"Meta-AAD: Active Anomaly Detection with Deep Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.07415","repositories_listed":1,"syntology":null},{"url":"/paper/text-generation-by-learning-from-off-policy","slug":"text-generation-by-learning-from-off-policy","title":"Text Generation by Learning from Demonstrations","date":"2020-09-16","arxiv_id":"2009.07839","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/text-generation-by-learning-from-off-policy#ran","syntology_url":"https://syntology.ai/paper/2009.07839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.07839"}},"official":{"repos":["yzpang/gold-off-policy-text-gen-iclr21"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/deep-actor-critic-learning-for-distributed","slug":"deep-actor-critic-learning-for-distributed","title":"Deep Actor-Critic Learning for Distributed Power Control in Wireless Mobile Networks","date":"2020-09-14","arxiv_id":"2009.06681","repositories_listed":1,"syntology":null},{"url":"/paper/vacsim-learning-effective-strategies-for","slug":"vacsim-learning-effective-strategies-for","title":"VacSIM: Learning Effective Strategies for COVID-19 Vaccine Distribution using Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.06602","repositories_listed":1,"syntology":null},{"url":"/paper/physically-embedded-planning-problems-new","slug":"physically-embedded-planning-problems-new","title":"Physically Embedded Planning Problems: New Challenges for Reinforcement Learning","date":"2020-09-11","arxiv_id":"2009.05524","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-optimal-frequency","slug":"reinforcement-learning-for-optimal-frequency","title":"Reinforcement Learning for Optimal Primary Frequency Control: A Lyapunov Approach","date":"2020-09-11","arxiv_id":"2009.05654","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-preserving-reinforcement-learning","slug":"semantic-preserving-reinforcement-learning","title":"Semantic-preserving Reinforcement Learning Attack Against Graph Neural Networks for Malware Detection","date":"2020-09-11","arxiv_id":"2009.05602","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-reinforcement-learning-with","slug":"a-framework-for-reinforcement-learning-with","title":"A framework for reinforcement learning with autocorrelated actions","date":"2020-09-10","arxiv_id":"2009.04777","repositories_listed":1,"syntology":null},{"url":"/paper/tripletree-a-versatile-interpretable","slug":"tripletree-a-versatile-interpretable","title":"TripleTree: A Versatile Interpretable Representation of Black Box Agents and their Environments","date":"2020-09-10","arxiv_id":"2009.04743","repositories_listed":1,"syntology":null},{"url":"/paper/dynode-neural-ordinary-differential-equations","slug":"dynode-neural-ordinary-differential-equations","title":"DyNODE: Neural Ordinary Differential Equations for Dynamics Modeling in Continuous Control","date":"2020-09-09","arxiv_id":"2009.04278","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynode-neural-ordinary-differential-equations#ran","syntology_url":"https://syntology.ai/paper/2009.04278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.04278"}},"official":{"repos":["vmartinezalvarez/DyNODE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-inverse-reinforcement-learning-for","slug":"bayesian-inverse-reinforcement-learning-for","title":"Bayesian Inverse Reinforcement Learning for Collective Animal Movement","date":"2020-09-08","arxiv_id":"2009.04003","repositories_listed":1,"syntology":null},{"url":"/paper/deep-active-inference-for-partially","slug":"deep-active-inference-for-partially","title":"Deep Active Inference for Partially Observable MDPs","date":"2020-09-08","arxiv_id":"2009.03622","repositories_listed":1,"syntology":null},{"url":"/paper/confuciux-autonomous-hardware-resource","slug":"confuciux-autonomous-hardware-resource","title":"ConfuciuX: Autonomous Hardware Resource Assignment for DNN Accelerators using Reinforcement Learning","date":"2020-09-04","arxiv_id":"2009.02010","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/confuciux-autonomous-hardware-resource#ran","syntology_url":"https://syntology.ai/paper/2009.02010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.02010"}},"official":{"repos":["maestro-project/confuciux"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/drle-decentralized-reinforcement-learning-at","slug":"drle-decentralized-reinforcement-learning-at","title":"DRLE: Decentralized Reinforcement Learning at the Edge for Traffic Light Control in the IoV","date":"2020-09-03","arxiv_id":"2009.01502","repositories_listed":1,"syntology":null},{"url":"/paper/optimality-based-analysis-of-xcsf-compaction","slug":"optimality-based-analysis-of-xcsf-compaction","title":"Optimality-based Analysis of XCSF Compaction in Discrete Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01476","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-automated-deep-reinforcement","slug":"sample-efficient-automated-deep-reinforcement","title":"Sample-Efficient Automated Deep Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01555","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-automated-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.01555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.01555"}},"official":{"repos":["automl/SEARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-model-based-stochastic-value-gradient","slug":"on-the-model-based-stochastic-value-gradient","title":"On the model-based stochastic value gradient for continuous reinforcement learning","date":"2020-08-28","arxiv_id":"2008.12775","repositories_listed":1,"syntology":null},{"url":"/paper/autofs-automated-feature-selection-via","slug":"autofs-automated-feature-selection-via","title":"AutoFS: Automated Feature Selection via Diversity-aware Interactive Reinforcement Learning","date":"2020-08-27","arxiv_id":"2008.12001","repositories_listed":1,"syntology":null},{"url":"/paper/query-focused-multi-document-summarisation-of","slug":"query-focused-multi-document-summarisation-of","title":"Query Focused Multi-document Summarisation of Biomedical Texts","date":"2020-08-27","arxiv_id":"2008.11986","repositories_listed":1,"syntology":null},{"url":"/paper/query-focused-multi-document-summarisation-of-1","slug":"query-focused-multi-document-summarisation-of-1","title":"Query Focused Multi-document Summarisation of Biomedical Texts: Macquarie Universiy and the Australian National University at BioASQ8b","date":"2020-08-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-off-policy-with-online-planning","slug":"learning-off-policy-with-online-planning","title":"Learning Off-Policy with Online Planning","date":"2020-08-23","arxiv_id":"2008.10066","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-off-policy-with-online-planning#ran","syntology_url":"https://syntology.ai/paper/2008.10066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.10066"}},"official":null}},{"url":"/paper/a-composable-specification-language-for-1","slug":"a-composable-specification-language-for-1","title":"A Composable Specification Language for Reinforcement Learning Tasks","date":"2020-08-21","arxiv_id":"2008.09293","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-composable-specification-language-for-1#ran","syntology_url":"https://syntology.ai/paper/2008.09293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09293"}},"official":{"repos":["keyshor/spectrl_tool"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/social-aware-incentive-mechanism-for","slug":"social-aware-incentive-mechanism-for","title":"Social-Aware Incentive Mechanism for VehicularCrowdsensing by Deep Reinforcement Learning","date":"2020-08-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/optimization-of-operation-parameters-towards","slug":"optimization-of-operation-parameters-towards","title":"Optimal control towards sustainable wastewater treatment plants based on multi-agent reinforcement learning","date":"2020-08-19","arxiv_id":"2008.10417","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-low-thrust","slug":"reinforcement-learning-for-low-thrust","title":"Reinforcement Learning for Low-Thrust Trajectory Design of Interplanetary Missions","date":"2020-08-19","arxiv_id":"2008.08501","repositories_listed":1,"syntology":null},{"url":"/paper/communicative-reinforcement-learning-agents","slug":"communicative-reinforcement-learning-agents","title":"Communicative Reinforcement Learning Agents for Landmark Detection in Brain Images","date":"2020-08-18","arxiv_id":"2008.08055","repositories_listed":1,"syntology":null},{"url":"/paper/learning-fair-policies-in-multiobjective-deep","slug":"learning-fair-policies-in-multiobjective-deep","title":"Learning Fair Policies in Multiobjective (Deep) Reinforcement Learning with Average and Discounted Rewards","date":"2020-08-18","arxiv_id":"2008.07773","repositories_listed":1,"syntology":null},{"url":"/paper/towards-closing-the-sim-to-real-gap-in","slug":"towards-closing-the-sim-to-real-gap-in","title":"Towards Closing the Sim-to-Real Gap in Collaborative Multi-Robot Deep Reinforcement Learning","date":"2020-08-18","arxiv_id":"2008.07875","repositories_listed":1,"syntology":null},{"url":"/paper/supersuit-simple-microwrappers-for","slug":"supersuit-simple-microwrappers-for","title":"SuperSuit: Simple Microwrappers for Reinforcement Learning Environments","date":"2020-08-17","arxiv_id":"2008.08932","repositories_listed":1,"syntology":null},{"url":"/paper/cautious-adaptation-for-reinforcement","slug":"cautious-adaptation-for-reinforcement","title":"Cautious Adaptation For Reinforcement Learning in Safety-Critical Settings","date":"2020-08-15","arxiv_id":"2008.06622","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-in-constrained","slug":"safe-reinforcement-learning-in-constrained","title":"Safe Reinforcement Learning in Constrained Markov Decision Processes","date":"2020-08-15","arxiv_id":"2008.06626","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-reinforcement-learning-in-constrained#ran","syntology_url":"https://syntology.ai/paper/2008.06626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06626"}},"official":{"repos":["akifumi-wachi-4/safe_near_optimal_mdp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-cross-entropy-method-for","slug":"sample-efficient-cross-entropy-method-for","title":"Sample-efficient Cross-Entropy Method for Real-time Planning","date":"2020-08-14","arxiv_id":"2008.06389","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sample-efficient-cross-entropy-method-for#ran","syntology_url":"https://syntology.ai/paper/2008.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06389"}},"official":{"repos":["martius-lab/iCEM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/batch-value-function-approximation-with-only","slug":"batch-value-function-approximation-with-only","title":"Batch Value-function Approximation with Only Realizability","date":"2020-08-11","arxiv_id":"2008.04990","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/batch-value-function-approximation-with-only#ran","syntology_url":"https://syntology.ai/paper/2008.04990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.04990"}},"official":null}},{"url":"/paper/safepilco-a-software-tool-for-safe-and-data","slug":"safepilco-a-software-tool-for-safe-and-data","title":"SafePILCO: a software tool for safe and data-efficient policy synthesis","date":"2020-08-07","arxiv_id":"2008.03273","repositories_listed":1,"syntology":null},{"url":"/paper/towards-sample-efficient-agents-through","slug":"towards-sample-efficient-agents-through","title":"Towards Sample Efficient Agents through Algorithmic Alignment","date":"2020-08-07","arxiv_id":"2008.03229","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-variational-model-based","slug":"contrastive-variational-model-based","title":"Contrastive Variational Reinforcement Learning for Complex Observations","date":"2020-08-06","arxiv_id":"2008.02430","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-tactile","slug":"deep-reinforcement-learning-for-tactile","title":"Deep Reinforcement Learning for Tactile Robotics: Learning to Type on a Braille Keyboard","date":"2020-08-06","arxiv_id":"2008.02646","repositories_listed":1,"syntology":null},{"url":"/paper/mixed-initiative-level-design-with-rl-brush","slug":"mixed-initiative-level-design-with-rl-brush","title":"Mixed-Initiative Level Design with RL Brush","date":"2020-08-06","arxiv_id":"2008.02778","repositories_listed":1,"syntology":null},{"url":"/paper/the-emergence-of-adversarial-communication-in","slug":"the-emergence-of-adversarial-communication-in","title":"The Emergence of Adversarial Communication in Multi-Agent Reinforcement Learning","date":"2020-08-06","arxiv_id":"2008.02616","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-emergence-of-adversarial-communication-in#ran","syntology_url":"https://syntology.ai/paper/2008.02616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02616"}},"official":{"repos":["proroklab/adversarial_comms"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-adaptive-task-offloading-in-edge","slug":"fast-adaptive-task-offloading-in-edge","title":"Fast Adaptive Task Offloading in Edge Computing based on Meta Reinforcement Learning","date":"2020-08-05","arxiv_id":"2008.02033","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-epidemic-control-saving-both-lives","slug":"reinforced-epidemic-control-saving-both-lives","title":"Reinforced Epidemic Control: Saving Both Lives and Economy","date":"2020-08-04","arxiv_id":"2008.01257","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-using","slug":"robust-reinforcement-learning-using","title":"Robust Reinforcement Learning using Adversarial Populations","date":"2020-08-04","arxiv_id":"2008.01825","repositories_listed":1,"syntology":null},{"url":"/paper/queueing-network-controls-via-deep","slug":"queueing-network-controls-via-deep","title":"Queueing Network Controls via Deep Reinforcement Learning","date":"2020-07-31","arxiv_id":"2008.01644","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/queueing-network-controls-via-deep#ran","syntology_url":"https://syntology.ai/paper/2008.01644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01644"}},"official":null}},{"url":"/paper/pixl2r-guiding-reinforcement-learning-using","slug":"pixl2r-guiding-reinforcement-learning-using","title":"PixL2R: Guiding Reinforcement Learning Using Natural Language by Mapping Pixels to Rewards","date":"2020-07-30","arxiv_id":"2007.15543","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pixl2r-guiding-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2007.15543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.15543"}},"official":{"repos":["prasoongoyal/PixL2R"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lifelong-incremental-reinforcement-learning","slug":"lifelong-incremental-reinforcement-learning","title":"Lifelong Incremental Reinforcement Learning with Online Bayesian Inference","date":"2020-07-28","arxiv_id":"2007.14196","repositories_listed":1,"syntology":null},{"url":"/paper/multi-step-reinforcement-learning-for-single","slug":"multi-step-reinforcement-learning-for-single","title":"Multi-Step Reinforcement Learning for Single Image Super-Resolution","date":"2020-07-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/combining-deep-reinforcement-learning-and-1","slug":"combining-deep-reinforcement-learning-and-1","title":"Combining Deep Reinforcement Learning and Search for Imperfect-Information Games","date":"2020-07-27","arxiv_id":"2007.13544","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/combining-deep-reinforcement-learning-and-1#ran","syntology_url":"https://syntology.ai/paper/2007.13544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.13544"}},"official":{"repos":["facebookresearch/rebel"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-discovery-of-local-rules-for","slug":"automated-discovery-of-local-rules-for","title":"Automated Discovery of Local Rules for Desired Collective-Level Behavior Through Reinforcement Learning","date":"2020-07-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/human-preference-scaling-with-demonstrations","slug":"human-preference-scaling-with-demonstrations","title":"Weak Human Preference Supervision For Deep Reinforcement Learning","date":"2020-07-25","arxiv_id":"2007.12904","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-exploration-under-uncertainty-via","slug":"autonomous-exploration-under-uncertainty-via","title":"Autonomous Exploration Under Uncertainty via Deep Reinforcement Learning on Graphs","date":"2020-07-24","arxiv_id":"2007.12640","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-robust-optimization-for-imitation","slug":"bayesian-robust-optimization-for-imitation","title":"Bayesian Robust Optimization for Imitation Learning","date":"2020-07-24","arxiv_id":"2007.12315","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayesian-robust-optimization-for-imitation#ran","syntology_url":"https://syntology.ai/paper/2007.12315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12315"}},"official":{"repos":["dsbrown1331/broil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clinician-in-the-loop-decision-making","slug":"clinician-in-the-loop-decision-making","title":"Clinician-in-the-Loop Decision Making: Reinforcement Learning with Near-Optimal Set-Valued Policies","date":"2020-07-24","arxiv_id":"2007.12678","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clinician-in-the-loop-decision-making#ran","syntology_url":"https://syntology.ai/paper/2007.12678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12678"}},"official":{"repos":["MLD3/RL-Set-Valued-Policy"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-inverse-reinforcement-learning-for","slug":"deep-inverse-reinforcement-learning-for","title":"Deep Inverse Reinforcement Learning for Structural Evolution of Small Molecules","date":"2020-07-24","arxiv_id":"2008.11804","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-with-3","slug":"distributional-reinforcement-learning-with-3","title":"Distributional Reinforcement Learning via Moment Matching","date":"2020-07-24","arxiv_id":"2007.12354","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distributional-reinforcement-learning-with-3#ran","syntology_url":"https://syntology.ai/paper/2007.12354","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12354"}},"official":{"repos":["thanhnguyentang/mmdrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/value-decomposition-multi-agent-actor-critics","slug":"value-decomposition-multi-agent-actor-critics","title":"Value-Decomposition Multi-Agent Actor-Critics","date":"2020-07-24","arxiv_id":"2007.12306","repositories_listed":1,"syntology":null},{"url":"/paper/challenging-common-bolus-advisor-for-self","slug":"challenging-common-bolus-advisor-for-self","title":"Challenging common bolus advisor for self-monitoring type-I diabetes patients using Reinforcement Learning","date":"2020-07-23","arxiv_id":"2007.11880","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-traffic-control-with-deep","slug":"adaptive-traffic-control-with-deep","title":"Adaptive Traffic Control with Deep Reinforcement Learning:Towards State-of-the-art and Beyond","date":"2020-07-21","arxiv_id":"2007.10960","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-deep-reinforcement-learning","slug":"integrating-deep-reinforcement-learning","title":"Integrating Deep Reinforcement Learning Networks with Health System Simulations","date":"2020-07-21","arxiv_id":"2008.07434","repositories_listed":1,"syntology":null},{"url":"/paper/battlesnake-challenge-a-multi-agent","slug":"battlesnake-challenge-a-multi-agent","title":"Battlesnake Challenge: A Multi-agent Reinforcement Learning Playground with Human-in-the-loop","date":"2020-07-20","arxiv_id":"2007.10504","repositories_listed":1,"syntology":null},{"url":"/paper/structure-mapping-for-transferability-of","slug":"structure-mapping-for-transferability-of","title":"Structure Mapping for Transferability of Causal Models","date":"2020-07-18","arxiv_id":"2007.09445","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/structure-mapping-for-transferability-of#ran","syntology_url":"https://syntology.ai/paper/2007.09445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.09445"}},"official":{"repos":["Information-Fusion-Lab-Umass/causal_transfer_learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/off-policy-reinforcement-learning-for","slug":"off-policy-reinforcement-learning-for","title":"Off-Policy Reinforcement Learning for Efficient and Effective GAN Architecture Search","date":"2020-07-17","arxiv_id":"2007.09180","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/off-policy-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2007.09180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.09180"}},"official":{"repos":["Yuantian013/E2GAN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/wordcraft-an-environment-for-benchmarking","slug":"wordcraft-an-environment-for-benchmarking","title":"WordCraft: An Environment for Benchmarking Commonsense Agents","date":"2020-07-17","arxiv_id":"2007.09185","repositories_listed":1,"syntology":null},{"url":"/paper/collision-avoidance-robotics-via-meta","slug":"collision-avoidance-robotics-via-meta","title":"Collision Avoidance Robotics Via Meta-Learning (CARML)","date":"2020-07-16","arxiv_id":"2007.08616","repositories_listed":1,"syntology":null},{"url":"/paper/provably-good-batch-reinforcement-learning","slug":"provably-good-batch-reinforcement-learning","title":"Provably Good Batch Reinforcement Learning Without Great Exploration","date":"2020-07-16","arxiv_id":"2007.08202","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provably-good-batch-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.08202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.08202"}},"official":null}},{"url":"/paper/weighing-counts-sequential-crowd-counting-by","slug":"weighing-counts-sequential-crowd-counting-by","title":"Weighing Counts: Sequential Crowd Counting by Reinforcement Learning","date":"2020-07-16","arxiv_id":"2007.08260","repositories_listed":1,"syntology":null},{"url":"/paper/developmental-reinforcement-learning-of","slug":"developmental-reinforcement-learning-of","title":"Developmental Reinforcement Learning of Control Policy of a Quadcopter UAV with Thrust Vectoring Rotors","date":"2020-07-15","arxiv_id":"2007.07793","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-reward-functions-using-anchor","slug":"identifying-reward-functions-using-anchor","title":"Deep PQR: Solving Inverse Reinforcement Learning using Anchor Actions","date":"2020-07-15","arxiv_id":"2007.07443","repositories_listed":1,"syntology":null},{"url":"/paper/single-partition-adaptive-q-learning","slug":"single-partition-adaptive-q-learning","title":"Single-partition adaptive Q-learning","date":"2020-07-14","arxiv_id":"2007.06741","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-of-musculoskeletal","slug":"reinforcement-learning-of-musculoskeletal","title":"Reinforcement Learning of Musculoskeletal Control from Functional Simulations","date":"2020-07-13","arxiv_id":"2007.06669","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-reinforcement-learning-with-1","slug":"data-efficient-reinforcement-learning-with-1","title":"Data-Efficient Reinforcement Learning with Self-Predictive Representations","date":"2020-07-12","arxiv_id":"2007.05929","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/data-efficient-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2007.05929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05929"}},"official":{"repos":["mila-iqia/spr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-abstract-models-for-strategic","slug":"learning-abstract-models-for-strategic","title":"Learning Abstract Models for Strategic Exploration and Fast Reward Transfer","date":"2020-07-12","arxiv_id":"2007.05896","repositories_listed":1,"syntology":null},{"url":"/paper/xcs-as-a-reinforcement-learning-approach-to","slug":"xcs-as-a-reinforcement-learning-approach-to","title":"XCS as a reinforcement learning approach to automatic test case prioritization","date":"2020-07-12","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"2a3cbf2042887217f65671b1eb094133b5ea4e027ad4939aa4e43106a868241c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}