{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/32","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":32,"pages_in_order":152,"rows_per_page":100,"rows":[3101,3200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/31","next":"/task/reinforcement-learning-1/papers/33","papers":[{"url":"/paper/sequence-adaptation-via-reinforcement","slug":"sequence-adaptation-via-reinforcement","title":"Sequence Adaptation via Reinforcement Learning in Recommender Systems","date":"2021-07-31","arxiv_id":"2108.01442","repositories_listed":1,"syntology":null},{"url":"/paper/strategically-efficient-exploration-in","slug":"strategically-efficient-exploration-in","title":"Strategically Efficient Exploration in Competitive Multi-agent Reinforcement Learning","date":"2021-07-30","arxiv_id":"2107.14698","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-epidemics-through-optimal","slug":"controlling-epidemics-through-optimal","title":"Controlling epidemics through optimal allocation of test kits and vaccine doses across networks","date":"2021-07-29","arxiv_id":"2107.13709","repositories_listed":1,"syntology":null},{"url":"/paper/tianshou-a-highly-modularized-deep","slug":"tianshou-a-highly-modularized-deep","title":"Tianshou: a Highly Modularized Deep Reinforcement Learning Library","date":"2021-07-29","arxiv_id":"2107.14171","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tianshou-a-highly-modularized-deep#ran","syntology_url":"https://syntology.ai/paper/2107.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14171"}},"official":{"repos":["thu-ml/tianshou"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finding-failures-in-high-fidelity-simulation","slug":"finding-failures-in-high-fidelity-simulation","title":"Finding Failures in High-Fidelity Simulation using Adaptive Stress Testing and the Backward Algorithm","date":"2021-07-27","arxiv_id":"2107.12940","repositories_listed":1,"syntology":null},{"url":"/paper/model-selection-for-offline-reinforcement","slug":"model-selection-for-offline-reinforcement","title":"Model Selection for Offline Reinforcement Learning: Practical Considerations for Healthcare Settings","date":"2021-07-23","arxiv_id":"2107.11003","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-quadratic-optimization-with","slug":"accelerating-quadratic-optimization-with","title":"Accelerating Quadratic Optimization with Reinforcement Learning","date":"2021-07-22","arxiv_id":"2107.10847","repositories_listed":1,"syntology":null},{"url":"/paper/demonstration-guided-reinforcement-learning","slug":"demonstration-guided-reinforcement-learning","title":"Demonstration-Guided Reinforcement Learning with Learned Skills","date":"2021-07-21","arxiv_id":"2107.10253","repositories_listed":1,"syntology":null},{"url":"/paper/toward-collaborative-reinforcement-learning","slug":"toward-collaborative-reinforcement-learning","title":"Toward Collaborative Reinforcement Learning Agents that Communicate Through Text-Based Natural Language","date":"2021-07-20","arxiv_id":"2107.09356","repositories_listed":1,"syntology":null},{"url":"/paper/a-sustainable-ecosystem-through-emergent","slug":"a-sustainable-ecosystem-through-emergent","title":"A Sustainable Ecosystem through Emergent Cooperation in Multi-Agent Reinforcement Learning","date":"2021-07-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-exploration-and-exploitation-in","slug":"decoupling-exploration-and-exploitation-in","title":"Decoupled Reinforcement Learning to Stabilise Intrinsically-Motivated Exploration","date":"2021-07-19","arxiv_id":"2107.08966","repositories_listed":1,"syntology":null},{"url":"/paper/reward-weighted-regression-converges-to-a","slug":"reward-weighted-regression-converges-to-a","title":"Reward-Weighted Regression Converges to a Global Optimum","date":"2021-07-19","arxiv_id":"2107.09088","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-weighted-regression-converges-to-a#ran","syntology_url":"https://syntology.ai/paper/2107.09088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.09088"}},"official":{"repos":["dylanashley/reward-weighted-regression"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/co-designing-intelligent-control-of-building","slug":"co-designing-intelligent-control-of-building","title":"Co-designing Intelligent Control of Building HVACs and Microgrids","date":"2021-07-18","arxiv_id":"2107.08378","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-autonomous-car-racing-using-deep","slug":"vision-based-autonomous-car-racing-using-deep","title":"Vision-Based Autonomous Car Racing Using Deep Imitative Reinforcement Learning","date":"2021-07-18","arxiv_id":"2107.08325","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-with-2","slug":"hierarchical-reinforcement-learning-with-2","title":"Hierarchical Reinforcement Learning with Optimal Level Synchronization based on a Deep Generative Model","date":"2021-07-17","arxiv_id":"2107.08183","repositories_listed":1,"syntology":null},{"url":"/paper/megaverse-simulating-embodied-agents-at-one","slug":"megaverse-simulating-embodied-agents-at-one","title":"Megaverse: Simulating Embodied Agents at One Million Experiences per Second","date":"2021-07-17","arxiv_id":"2107.08170","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-environment-for-2","slug":"a-reinforcement-learning-environment-for-2","title":"A Reinforcement Learning Environment for Mathematical Reasoning via Program Synthesis","date":"2021-07-15","arxiv_id":"2107.07373","repositories_listed":1,"syntology":null},{"url":"/paper/pc-mlp-model-based-reinforcement-learning","slug":"pc-mlp-model-based-reinforcement-learning","title":"PC-MLP: Model-based Reinforcement Learning with Policy Cover Guided Exploration","date":"2021-07-15","arxiv_id":"2107.07410","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pc-mlp-model-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2107.07410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07410"}},"official":{"repos":["yudasong/PCMLP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/centralized-model-and-exploration-policy-for","slug":"centralized-model-and-exploration-policy-for","title":"Centralized Model and Exploration Policy for Multi-Agent RL","date":"2021-07-14","arxiv_id":"2107.06434","repositories_listed":1,"syntology":null},{"url":"/paper/deep-adaptive-multi-intention-inverse","slug":"deep-adaptive-multi-intention-inverse","title":"Deep Adaptive Multi-Intention Inverse Reinforcement Learning","date":"2021-07-14","arxiv_id":"2107.06692","repositories_listed":1,"syntology":null},{"url":"/paper/safer-reinforcement-learning-through","slug":"safer-reinforcement-learning-through","title":"Safer Reinforcement Learning through Transferable Instinct Networks","date":"2021-07-14","arxiv_id":"2107.06686","repositories_listed":1,"syntology":null},{"url":"/paper/surgical-instruction-generation-with","slug":"surgical-instruction-generation-with","title":"Surgical Instruction Generation with Transformers","date":"2021-07-14","arxiv_id":"2107.06964","repositories_listed":1,"syntology":null},{"url":"/paper/carle-s-game-an-open-ended-challenge-in","slug":"carle-s-game-an-open-ended-challenge-in","title":"Carle's Game: An Open-Ended Challenge in Exploratory Machine Creativity","date":"2021-07-13","arxiv_id":"2107.05786","repositories_listed":1,"syntology":null},{"url":"/paper/rellie-deep-reinforcement-learning-for","slug":"rellie-deep-reinforcement-learning-for","title":"ReLLIE: Deep Reinforcement Learning for Customized Low-Light Image Enhancement","date":"2021-07-13","arxiv_id":"2107.05830","repositories_listed":1,"syntology":null},{"url":"/paper/shortest-path-constrained-reinforcement-1","slug":"shortest-path-constrained-reinforcement-1","title":"Shortest-Path Constrained Reinforcement Learning for Sparse Reward Tasks","date":"2021-07-13","arxiv_id":"2107.06405","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shortest-path-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2107.06405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06405"}},"official":{"repos":["srsohn/shortest-path-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-constraining-in-weight-space-for","slug":"behavior-constraining-in-weight-space-for","title":"Behavior Constraining in Weight Space for Offline Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.05479","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-offline-distributional","slug":"conservative-offline-distributional","title":"Conservative Offline Distributional Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.06106","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-offline-distributional#ran","syntology_url":"https://syntology.ai/paper/2107.06106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06106"}},"official":{"repos":["JasonMa2016/CODAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explore-and-control-with-adversarial-surprise","slug":"explore-and-control-with-adversarial-surprise","title":"Explore and Control with Adversarial Surprise","date":"2021-07-12","arxiv_id":"2107.07394","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explore-and-control-with-adversarial-surprise#ran","syntology_url":"https://syntology.ai/paper/2107.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07394"}},"official":null}},{"url":"/paper/modeling-explicit-concerning-states-for","slug":"modeling-explicit-concerning-states-for","title":"Modeling Explicit Concerning States for Reinforcement Learning in Visual Dialogue","date":"2021-07-12","arxiv_id":"2107.05250","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-explicit-concerning-states-for#ran","syntology_url":"https://syntology.ai/paper/2107.05250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.05250"}},"official":{"repos":["zipengxuc/ecs-visdial-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-better-laplacian-representation-in","slug":"towards-better-laplacian-representation-in","title":"Towards Better Laplacian Representation in Reinforcement Learning with Generalized Graph Drawing","date":"2021-07-12","arxiv_id":"2107.05545","repositories_listed":1,"syntology":null},{"url":"/paper/backprop-free-reinforcement-learning-with","slug":"backprop-free-reinforcement-learning-with","title":"Backprop-Free Reinforcement Learning with Active Neural Generative Coding","date":"2021-07-10","arxiv_id":"2107.07046","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/backprop-free-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2107.07046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07046"}},"official":{"repos":["ago109/active-neural-generative-coding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ls3-latent-space-safe-sets-for-long-horizon","slug":"ls3-latent-space-safe-sets-for-long-horizon","title":"LS3: Latent Space Safe Sets for Long-Horizon Visuomotor Control of Sparse Reward Iterative Tasks","date":"2021-07-10","arxiv_id":"2107.04775","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-an-optical-interferometer-with-beam","slug":"aligning-an-optical-interferometer-with-beam","title":"Aligning an optical interferometer with beam divergence control and continuous action space","date":"2021-07-09","arxiv_id":"2107.04457","repositories_listed":1,"syntology":null},{"url":"/paper/bayessimig-scalable-parameter-inference-for","slug":"bayessimig-scalable-parameter-inference-for","title":"BayesSimIG: Scalable Parameter Inference for Adaptive Domain Randomization with IsaacGym","date":"2021-07-09","arxiv_id":"2107.04527","repositories_listed":1,"syntology":null},{"url":"/paper/learning-interaction-aware-guidance-policies","slug":"learning-interaction-aware-guidance-policies","title":"Learning Interaction-aware Guidance Policies for Motion Planning in Dense Traffic Scenarios","date":"2021-07-09","arxiv_id":"2107.04538","repositories_listed":1,"syntology":null},{"url":"/paper/computational-benefits-of-intermediate","slug":"computational-benefits-of-intermediate","title":"Computational Benefits of Intermediate Rewards for Goal-Reaching Policy Learning","date":"2021-07-08","arxiv_id":"2107.03961","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/computational-benefits-of-intermediate#ran","syntology_url":"https://syntology.ai/paper/2107.03961","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03961"}},"official":{"repos":["kebaek/minigrid"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-vision-guided-quadrupedal-locomotion","slug":"learning-vision-guided-quadrupedal-locomotion","title":"Learning Vision-Guided Quadrupedal Locomotion End-to-End with Cross-Modal Transformers","date":"2021-07-08","arxiv_id":"2107.03996","repositories_listed":1,"syntology":null},{"url":"/paper/offline-meta-reinforcement-learning-with-1","slug":"offline-meta-reinforcement-learning-with-1","title":"Offline Meta-Reinforcement Learning with Online Self-Supervision","date":"2021-07-08","arxiv_id":"2107.03974","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-online-service-coordination-using","slug":"distributed-online-service-coordination-using","title":"Distributed Online Service Coordination Using Deep Reinforcement Learning","date":"2021-07-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adarl-what-where-and-how-to-adapt-in-transfer","slug":"adarl-what-where-and-how-to-adapt-in-transfer","title":"AdaRL: What, Where, and How to Adapt in Transfer Reinforcement Learning","date":"2021-07-06","arxiv_id":"2107.02729","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/adarl-what-where-and-how-to-adapt-in-transfer#ran","syntology_url":"https://syntology.ai/paper/2107.02729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.02729"}},"official":{"repos":["adaptive-rl/adarl-code"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-mutual-information-mummi-training","slug":"multi-modal-mutual-information-mummi-training","title":"Multi-Modal Mutual Information (MuMMI) Training for Robust Self-Supervised Deep Reinforcement Learning","date":"2021-07-06","arxiv_id":"2107.02339","repositories_listed":1,"syntology":null},{"url":"/paper/the-sjtu-system-for-dcase2021-challenge-task","slug":"the-sjtu-system-for-dcase2021-challenge-task","title":"THE SJTU SYSTEM FOR DCASE2021 CHALLENGE TASK 6: AUDIO CAPTIONING BASED ON ENCODER PRE-TRAINING AND REINFORCEMENT LEARNING","date":"2021-07-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/agents-that-listen-high-throughput","slug":"agents-that-listen-high-throughput","title":"Agents that Listen: High-Throughput Reinforcement Learning with Multiple Sensory Systems","date":"2021-07-05","arxiv_id":"2107.02195","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-and-auxiliary-tasks-for-data","slug":"ensemble-and-auxiliary-tasks-for-data","title":"Ensemble and Auxiliary Tasks for Data-Efficient Deep Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01904","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ensemble-and-auxiliary-tasks-for-data#ran","syntology_url":"https://syntology.ai/paper/2107.01904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.01904"}},"official":{"repos":["NUS-LID/RENAULT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imputation-free-learning-from-incomplete","slug":"imputation-free-learning-from-incomplete","title":"Gradient Importance Learning for Incomplete Observations","date":"2021-07-05","arxiv_id":"2107.01983","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-via-2","slug":"sample-efficient-reinforcement-learning-via-2","title":"Sample Efficient Reinforcement Learning via Model-Ensemble Exploration and Exploitation","date":"2021-07-05","arxiv_id":"2107.01825","repositories_listed":1,"syntology":null},{"url":"/paper/mava-a-research-framework-for-distributed","slug":"mava-a-research-framework-for-distributed","title":"Mava: a research library for distributed multi-agent reinforcement learning in JAX","date":"2021-07-03","arxiv_id":"2107.01460","repositories_listed":1,"syntology":null},{"url":"/paper/where-is-the-grass-greener-revisiting","slug":"where-is-the-grass-greener-revisiting","title":"Optimality Inductive Biases and Agnostic Guidelines for Offline Reinforcement Learning","date":"2021-07-03","arxiv_id":"2107.01407","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-kalman-filter-enkf-for-reinforcement","slug":"ensemble-kalman-filter-enkf-for-reinforcement","title":"Controlled Interacting Particle Algorithms for Simulation-based Reinforcement Learning","date":"2021-07-02","arxiv_id":"2107.01244","repositories_listed":1,"syntology":null},{"url":"/paper/rl-ncs-reinforcement-learning-based-data","slug":"rl-ncs-reinforcement-learning-based-data","title":"RL-NCS: Reinforcement learning based data-driven approach for nonuniform compressed sensing","date":"2021-07-02","arxiv_id":"2107.00838","repositories_listed":1,"syntology":null},{"url":"/paper/systematic-evaluation-of-causal-discovery-in-1","slug":"systematic-evaluation-of-causal-discovery-in-1","title":"Systematic Evaluation of Causal Discovery in Visual Model Based Reinforcement Learning","date":"2021-07-02","arxiv_id":"2107.00848","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/systematic-evaluation-of-causal-discovery-in-1#ran","syntology_url":"https://syntology.ai/paper/2107.00848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00848"}},"official":{"repos":["dido1998/CausalMBRL"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/distilling-reinforcement-learning-tricks-for","slug":"distilling-reinforcement-learning-tricks-for","title":"Distilling Reinforcement Learning Tricks for Video Games","date":"2021-07-01","arxiv_id":"2107.00703","repositories_listed":1,"syntology":null},{"url":"/paper/offline-to-online-reinforcement-learning-via","slug":"offline-to-online-reinforcement-learning-via","title":"Offline-to-Online Reinforcement Learning via Balanced Replay and Pessimistic Q-Ensemble","date":"2021-07-01","arxiv_id":"2107.00591","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-abstractive","slug":"reinforcement-learning-for-abstractive","title":"Reinforcement Learning for Abstractive Question Summarization with Question-aware Semantic Rewards","date":"2021-07-01","arxiv_id":"2107.00176","repositories_listed":1,"syntology":null},{"url":"/paper/experience-driven-pcg-via-reinforcement","slug":"experience-driven-pcg-via-reinforcement","title":"Experience-Driven PCG via Reinforcement Learning: A Super Mario Bros Study","date":"2021-06-30","arxiv_id":"2106.15877","repositories_listed":1,"syntology":null},{"url":"/paper/koopman-spectrum-nonlinear-regulator-and","slug":"koopman-spectrum-nonlinear-regulator-and","title":"Koopman Spectrum Nonlinear Regulators and Efficient Online Learning","date":"2021-06-30","arxiv_id":"2106.15775","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-adversarial-attacks-on-1","slug":"understanding-adversarial-attacks-on-1","title":"Understanding Adversarial Attacks on Observations in Deep Reinforcement Learning","date":"2021-06-30","arxiv_id":"2106.15860","repositories_listed":1,"syntology":null},{"url":"/paper/a-convergent-and-efficient-deep-q-network","slug":"a-convergent-and-efficient-deep-q-network","title":"Convergent and Efficient Deep Q Network Algorithm","date":"2021-06-29","arxiv_id":"2106.15419","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-and-control-of-a-planar-quadrotor","slug":"analysis-and-control-of-a-planar-quadrotor","title":"Analysis and Control of a Planar Quadrotor","date":"2021-06-29","arxiv_id":"2106.15134","repositories_listed":1,"syntology":null},{"url":"/paper/globally-optimal-hierarchical-reinforcement","slug":"globally-optimal-hierarchical-reinforcement","title":"Globally Optimal Hierarchical Reinforcement Learning for Linearly-Solvable Markov Decision Processes","date":"2021-06-29","arxiv_id":"2106.15380","repositories_listed":1,"syntology":null},{"url":"/paper/learning-task-informed-abstraction","slug":"learning-task-informed-abstraction","title":"Learning Task Informed Abstractions","date":"2021-06-29","arxiv_id":"2106.15612","repositories_listed":1,"syntology":null},{"url":"/paper/causal-reinforcement-learning-using","slug":"causal-reinforcement-learning-using","title":"Causal Reinforcement Learning using Observational and Interventional Data","date":"2021-06-28","arxiv_id":"2106.14421","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-curriculum-learning-in-a-complex","slug":"multi-task-curriculum-learning-in-a-complex","title":"Multi-task curriculum learning in a complex, visual, hard-exploration domain: Minecraft","date":"2021-06-28","arxiv_id":"2106.14876","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-curriculum-learning-in-a-complex#ran","syntology_url":"https://syntology.ai/paper/2106.14876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14876"}},"official":null}},{"url":"/paper/graph-convolutional-memory-for-deep","slug":"graph-convolutional-memory-for-deep","title":"Graph Convolutional Memory using Topological Priors","date":"2021-06-27","arxiv_id":"2106.14117","repositories_listed":1,"syntology":null},{"url":"/paper/autopipeline-synthesize-data-pipelines-by","slug":"autopipeline-synthesize-data-pipelines-by","title":"Auto-Pipeline: Synthesizing Complex Data Pipelines By-Target Using Reinforcement Learning and Search","date":"2021-06-25","arxiv_id":"2106.13861","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-reinforcement-learning-from","slug":"compositional-reinforcement-learning-from","title":"Compositional Reinforcement Learning from Logical Specifications","date":"2021-06-25","arxiv_id":"2106.13906","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compositional-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2106.13906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13906"}},"official":{"repos":["keyshor/dirl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-goal-reinforcement-learning","slug":"multi-goal-reinforcement-learning","title":"Multi-Goal Reinforcement Learning environments for simulated Franka Emika Panda robot","date":"2021-06-25","arxiv_id":"2106.13687","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-goal-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.13687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13687"}},"official":{"repos":["qgallouedec/panda-gym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/brax-a-differentiable-physics-engine-for","slug":"brax-a-differentiable-physics-engine-for","title":"Brax -- A Differentiable Physics Engine for Large Scale Rigid Body Simulation","date":"2021-06-24","arxiv_id":"2106.13281","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/brax-a-differentiable-physics-engine-for#ran","syntology_url":"https://syntology.ai/paper/2106.13281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13281"}},"official":{"repos":["google/brax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-reinforcement-learning-via-latent-1","slug":"model-based-reinforcement-learning-via-latent-1","title":"Model-Based Reinforcement Learning via Latent-Space Collocation","date":"2021-06-24","arxiv_id":"2106.13229","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-gradient-estimators-for-meta","slug":"unifying-gradient-estimators-for-meta","title":"Unifying Gradient Estimators for Meta-Reinforcement Learning via Off-Policy Evaluation","date":"2021-06-24","arxiv_id":"2106.13125","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-gradient-estimators-for-meta#ran","syntology_url":"https://syntology.ai/paper/2106.13125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13125"}},"official":{"repos":["robintyh1/neurips2021-meta-gradient-offpolicy-evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bregman-gradient-policy-optimization","slug":"bregman-gradient-policy-optimization","title":"Bregman Gradient Policy Optimization","date":"2021-06-23","arxiv_id":"2106.12112","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bregman-gradient-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2106.12112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.12112"}},"official":{"repos":["gaosh/bgpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-dialogue-guided","slug":"reinforcement-learning-based-dialogue-guided","title":"Reinforcement Learning-based Dialogue Guided Event Extraction to Exploit Argument Relations","date":"2021-06-23","arxiv_id":"2106.12384","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-phy-layer","slug":"reinforcement-learning-for-phy-layer","title":"Reinforcement Learning for Physical Layer Communications","date":"2021-06-22","arxiv_id":"2106.11595","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-heuristic-multi-agent-path","slug":"distributed-heuristic-multi-agent-path","title":"Distributed Heuristic Multi-Agent Path Finding with Communication","date":"2021-06-21","arxiv_id":"2106.11365","repositories_listed":1,"syntology":null},{"url":"/paper/optidice-offline-policy-optimization-via","slug":"optidice-offline-policy-optimization-via","title":"OptiDICE: Offline Policy Optimization via Stationary Distribution Correction Estimation","date":"2021-06-21","arxiv_id":"2106.10783","repositories_listed":1,"syntology":null},{"url":"/paper/a-max-min-entropy-framework-for-reinforcement","slug":"a-max-min-entropy-framework-for-reinforcement","title":"A Max-Min Entropy Framework for Reinforcement Learning","date":"2021-06-19","arxiv_id":"2106.10517","repositories_listed":1,"syntology":null},{"url":"/paper/made-exploration-via-maximizing-deviation","slug":"made-exploration-via-maximizing-deviation","title":"MADE: Exploration via Maximizing Deviation from Explored Regions","date":"2021-06-18","arxiv_id":"2106.10268","repositories_listed":1,"syntology":null},{"url":"/paper/proper-value-equivalence","slug":"proper-value-equivalence","title":"Proper Value Equivalence","date":"2021-06-18","arxiv_id":"2106.10316","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-reinforcement-learning-via-1","slug":"towards-safe-reinforcement-learning-via-1","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value at Risk","date":"2021-06-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-demonstration-without","slug":"learning-from-demonstration-without","title":"Learning from Demonstration without Demonstrations","date":"2021-06-17","arxiv_id":"2106.09203","repositories_listed":1,"syntology":null},{"url":"/paper/secant-self-expert-cloning-for-zero-shot","slug":"secant-self-expert-cloning-for-zero-shot","title":"SECANT: Self-Expert Cloning for Zero-Shot Generalization of Visual Policies","date":"2021-06-17","arxiv_id":"2106.09678","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/secant-self-expert-cloning-for-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2106.09678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09678"}},"official":{"repos":["LinxiFan/SECANT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-rl-without-off-policy-evaluation","slug":"offline-rl-without-off-policy-evaluation","title":"Offline RL Without Off-Policy Evaluation","date":"2021-06-16","arxiv_id":"2106.08909","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-attacks-against-deep-reinforcement","slug":"real-time-attacks-against-deep-reinforcement","title":"Real-time Adversarial Perturbations against Deep Reinforcement Learning Policies: Attacks and Defenses","date":"2021-06-16","arxiv_id":"2106.08746","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-attacks-against-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.08746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08746"}},"official":{"repos":["ssg-research/ad3-action-distribution-divergence-detector"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-weaknesses-of-reinforcement-1","slug":"revisiting-the-weaknesses-of-reinforcement-1","title":"Revisiting the Weaknesses of Reinforcement Learning for Neural Machine Translation","date":"2021-06-16","arxiv_id":"2106.08942","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-weaknesses-of-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2106.08942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08942"}},"official":{"repos":["samuki/reinforce-joey"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-using-advantage","slug":"safe-reinforcement-learning-using-advantage","title":"Safe Reinforcement Learning Using Advantage-Based Intervention","date":"2021-06-16","arxiv_id":"2106.09110","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-reinforcement-learning-using-advantage#ran","syntology_url":"https://syntology.ai/paper/2106.09110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09110"}},"official":{"repos":["nolanwagener/safe_rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-continuous-control-with-episodic","slug":"solving-continuous-control-with-episodic","title":"Solving Continuous Control with Episodic Memory","date":"2021-06-16","arxiv_id":"2106.08832","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-conservation","slug":"deep-reinforcement-learning-for-conservation","title":"Deep Reinforcement Learning for Conservation Decisions","date":"2021-06-15","arxiv_id":"2106.08272","repositories_listed":1,"syntology":null},{"url":"/paper/randomized-exploration-for-reinforcement","slug":"randomized-exploration-for-reinforcement","title":"Randomized Exploration for Reinforcement Learning with General Value Function Approximation","date":"2021-06-15","arxiv_id":"2106.07841","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/randomized-exploration-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.07841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07841"}},"official":{"repos":["qlan3/Explorer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-under-minimax","slug":"robust-reinforcement-learning-under-minimax","title":"Robust Reinforcement Learning Under Minimax Regret for Green Security","date":"2021-06-15","arxiv_id":"2106.08413","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-reinforcement-learning-under-minimax#ran","syntology_url":"https://syntology.ai/paper/2106.08413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08413"}},"official":{"repos":["lily-x/mirror"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rsoccer-a-framework-for-studying","slug":"rsoccer-a-framework-for-studying","title":"rSoccer: A Framework for Studying Reinforcement Learning in Small and Very Small Size Robot Soccer","date":"2021-06-15","arxiv_id":"2106.12895","repositories_listed":1,"syntology":null},{"url":"/paper/learning-intrusion-prevention-policies","slug":"learning-intrusion-prevention-policies","title":"Learning Intrusion Prevention Policies through Optimal Stopping","date":"2021-06-14","arxiv_id":"2106.07160","repositories_listed":1,"syntology":null},{"url":"/paper/text-generation-with-efficient-soft-q","slug":"text-generation-with-efficient-soft-q","title":"Efficient (Soft) Q-Learning for Text Generation with Limited Good Data","date":"2021-06-14","arxiv_id":"2106.07704","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/text-generation-with-efficient-soft-q#ran","syntology_url":"https://syntology.ai/paper/2106.07704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07704"}},"official":{"repos":["HanGuo97/soft-Q-learning-for-text-generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contingency-aware-influence-maximization-a","slug":"contingency-aware-influence-maximization-a","title":"Contingency-Aware Influence Maximization: A Reinforcement Learning Approach","date":"2021-06-13","arxiv_id":"2106.07039","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-group","slug":"deep-reinforcement-learning-based-group","title":"Deep Reinforcement Learning based Group Recommender System","date":"2021-06-13","arxiv_id":"2106.06900","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-as-one-big-sequence-1","slug":"reinforcement-learning-as-one-big-sequence-1","title":"Reinforcement Learning as One Big Sequence Modeling Problem","date":"2021-06-13","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-to-4","slug":"a-deep-reinforcement-learning-approach-to-4","title":"A Deep Reinforcement Learning Approach to Marginalized Importance Sampling with the Successor Representation","date":"2021-06-12","arxiv_id":"2106.06854","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-deep-reinforcement-learning-approach-to-4#ran","syntology_url":"https://syntology.ai/paper/2106.06854","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06854"}},"official":{"repos":["sfujim/SR-DICE"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-game-theoretic-approach-to-multi-agent","slug":"a-game-theoretic-approach-to-multi-agent","title":"A Game-Theoretic Approach to Multi-Agent Trust Region Optimization","date":"2021-06-12","arxiv_id":"2106.06828","repositories_listed":1,"syntology":null},{"url":"/paper/douzero-mastering-doudizhu-with-self-play","slug":"douzero-mastering-doudizhu-with-self-play","title":"DouZero: Mastering DouDizhu with Self-Play Deep Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06135","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/douzero-mastering-doudizhu-with-self-play#ran","syntology_url":"https://syntology.ai/paper/2106.06135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06135"}},"official":{"repos":["kwai/DouZero"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/wax-ml-a-python-library-for-machine-learning","slug":"wax-ml-a-python-library-for-machine-learning","title":"WAX-ML: A Python library for machine learning and feedback loops on streaming data","date":"2021-06-11","arxiv_id":"2106.06524","repositories_listed":1,"syntology":null},{"url":"/paper/ermas-becoming-robust-to-reward-function-sim","slug":"ermas-becoming-robust-to-reward-function-sim","title":"Learning to Play General-Sum Games Against Multiple Boundedly Rational Agents","date":"2021-06-10","arxiv_id":"2106.05492","repositories_listed":1,"syntology":null}],"record_sha256":"e8bc15c80c3715fdc712484c8bf03e16f8956d6afb8d40181c0aeffcec88e173","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}