{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/40","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":40,"pages_in_order":152,"rows_per_page":100,"rows":[3901,4000],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/39","next":"/task/reinforcement-learning-1/papers/41","papers":[{"url":"/paper/mvp-unified-motion-and-visual-self-supervised","slug":"mvp-unified-motion-and-visual-self-supervised","title":"MVP: Unified Motion and Visual Self-Supervised Learning for Large-Scale Robotic Navigation","date":"2020-03-02","arxiv_id":"2003.00667","repositories_listed":1,"syntology":null},{"url":"/paper/ppmc-training-algorithm-a-robot-independent","slug":"ppmc-training-algorithm-a-robot-independent","title":"PPMC RL Training Algorithm: Rough Terrain Intelligent Robots through Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.02655","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-stochastic-policy-gradient-algorithm","slug":"a-hybrid-stochastic-policy-gradient-algorithm","title":"A Hybrid Stochastic Policy Gradient Algorithm for Reinforcement Learning","date":"2020-03-01","arxiv_id":"2003.00430","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hybrid-stochastic-policy-gradient-algorithm#ran","syntology_url":"https://syntology.ai/paper/2003.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00430"}},"official":{"repos":["unc-optimization/ProxHSPGA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-catastrophic-interference-in-atari-2600","slug":"on-catastrophic-interference-in-atari-2600","title":"On Catastrophic Interference in Atari 2600 Games","date":"2020-02-28","arxiv_id":"2002.12499","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-robotic-nanofabrication-with","slug":"autonomous-robotic-nanofabrication-with","title":"Autonomous robotic nanofabrication with reinforcement learning","date":"2020-02-27","arxiv_id":"2002.11952","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-of-risk-constrained","slug":"reinforcement-learning-of-risk-constrained","title":"Reinforcement Learning of Risk-Constrained Policies in Markov Decision Processes","date":"2020-02-27","arxiv_id":"2002.12086","repositories_listed":1,"syntology":null},{"url":"/paper/training-adversarial-agents-to-exploit","slug":"training-adversarial-agents-to-exploit","title":"Training Adversarial Agents to Exploit Weaknesses in Deep Control Policies","date":"2020-02-27","arxiv_id":"2002.12078","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-control-for","slug":"efficient-reinforcement-learning-control-for","title":"Efficient reinforcement learning control for continuum robots based on Inexplicit Prior Knowledge","date":"2020-02-26","arxiv_id":"2002.11573","repositories_listed":1,"syntology":null},{"url":"/paper/mid-flight-propeller-failure-detection-and","slug":"mid-flight-propeller-failure-detection-and","title":"Mid-flight Propeller Failure Detection and Control of Propeller-deficient Quadcopter using Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11564","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-exploration-even-with-a-1#ran","syntology_url":"https://syntology.ai/paper/2002.12174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.12174"}},"official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-reinforcement-learning-in-the","slug":"using-reinforcement-learning-in-the","title":"Using Reinforcement Learning in the Algorithmic Trading Problem","date":"2020-02-26","arxiv_id":"2002.11523","repositories_listed":1,"syntology":null},{"url":"/paper/human-apprenticeship-learning-via-kernel","slug":"human-apprenticeship-learning-via-kernel","title":"Reward Shaping for Human Learning via Inverse Reinforcement Learning","date":"2020-02-25","arxiv_id":"2002.10904","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-deep-reinforcement-learning-with","slug":"off-policy-deep-reinforcement-learning-with","title":"Off-Policy Deep Reinforcement Learning with Analogous Disentangled Exploration","date":"2020-02-25","arxiv_id":"2002.10738","repositories_listed":1,"syntology":null},{"url":"/paper/rewriting-history-with-inverse-rl-hindsight","slug":"rewriting-history-with-inverse-rl-hindsight","title":"Rewriting History with Inverse RL: Hindsight Inference for Policy Improvement","date":"2020-02-25","arxiv_id":"2002.11089","repositories_listed":1,"syntology":null},{"url":"/paper/whole-body-control-of-a-mobile-manipulator","slug":"whole-body-control-of-a-mobile-manipulator","title":"Whole-Body Control of a Mobile Manipulator using End-to-End Reinforcement Learning","date":"2020-02-25","arxiv_id":"2003.02637","repositories_listed":1,"syntology":null},{"url":"/paper/reconfigurable-intelligent-surface-assisted","slug":"reconfigurable-intelligent-surface-assisted","title":"Reconfigurable Intelligent Surface Assisted Multiuser MISO Systems Exploiting Deep Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10072","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-for-probabilistic","slug":"safe-reinforcement-learning-for-probabilistic","title":"Safe reinforcement learning for probabilistic reachability and safety specifications: A Lyapunov-based approach","date":"2020-02-24","arxiv_id":"2002.10126","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-particle-filter-reinforcement-1","slug":"discriminative-particle-filter-reinforcement-1","title":"Discriminative Particle Filter Reinforcement Learning for Complex Partial Observations","date":"2020-02-23","arxiv_id":"2002.09884","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-framework-for-deep","slug":"reinforcement-learning-framework-for-deep","title":"Reinforcement Learning Framework for Deep Brain Stimulation Study","date":"2020-02-22","arxiv_id":"2002.10948","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-through","slug":"efficient-deep-reinforcement-learning-through","title":"Efficient Deep Reinforcement Learning via Adaptive Policy Transfer","date":"2020-02-19","arxiv_id":"2002.08037","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-avoid-being-eaten-by-a-grue","slug":"how-to-avoid-being-eaten-by-a-grue","title":"How To Avoid Being Eaten By a Grue: Exploration Strategies for Text-Adventure Agents","date":"2020-02-19","arxiv_id":"2002.08795","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-avoid-being-eaten-by-a-grue#ran","syntology_url":"https://syntology.ai/paper/2002.08795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08795"}},"official":{"repos":["rajammanabrolu/Q-BERT"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/sim2real-transfer-for-reinforcement-learning","slug":"sim2real-transfer-for-reinforcement-learning","title":"Sim2Real Transfer for Reinforcement Learning without Dynamics Randomization","date":"2020-02-19","arxiv_id":"2002.11635","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-estimator-selection-for-off-policy","slug":"adaptive-estimator-selection-for-off-policy","title":"Adaptive Estimator Selection for Off-Policy Evaluation","date":"2020-02-18","arxiv_id":"2002.07729","repositories_listed":1,"syntology":null},{"url":"/paper/generating-automatic-curricula-via-self","slug":"generating-automatic-curricula-via-self","title":"Generating Automatic Curricula via Self-Supervised Active Domain Randomization","date":"2020-02-18","arxiv_id":"2002.07911","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-automatic-curricula-via-self#ran","syntology_url":"https://syntology.ai/paper/2002.07911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.07911"}},"official":null}},{"url":"/paper/reinforcement-learning-for-molecular-design","slug":"reinforcement-learning-for-molecular-design","title":"Reinforcement Learning for Molecular Design Guided by Quantum Mechanics","date":"2020-02-18","arxiv_id":"2002.07717","repositories_listed":1,"syntology":null},{"url":"/paper/control-frequency-adaptation-via-action","slug":"control-frequency-adaptation-via-action","title":"Control Frequency Adaptation via Action Persistence in Batch Reinforcement Learning","date":"2020-02-17","arxiv_id":"2002.06836","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/control-frequency-adaptation-via-action#ran","syntology_url":"https://syntology.ai/paper/2002.06836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06836"}},"official":{"repos":["albertometelli/pfqi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kalman-meets-bellman-improving-policy","slug":"kalman-meets-bellman-improving-policy","title":"Kalman meets Bellman: Improving Policy Evaluation through Value Tracking","date":"2020-02-17","arxiv_id":"2002.07171","repositories_listed":1,"syntology":null},{"url":"/paper/r-maddpg-for-partially-observable","slug":"r-maddpg-for-partially-observable","title":"R-MADDPG for Partially Observable Environments and Limited Communication","date":"2020-02-16","arxiv_id":"2002.06684","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-active-learning-for-image-1","slug":"reinforced-active-learning-for-image-1","title":"Reinforced active learning for image segmentation","date":"2020-02-16","arxiv_id":"2002.06583","repositories_listed":1,"syntology":null},{"url":"/paper/deep-rl-agent-for-a-real-time-action-strategy","slug":"deep-rl-agent-for-a-real-time-action-strategy","title":"Deep RL Agent for a Real-Time Action Strategy Game","date":"2020-02-15","arxiv_id":"2002.06290","repositories_listed":1,"syntology":null},{"url":"/paper/universal-value-density-estimation-for","slug":"universal-value-density-estimation-for","title":"Universal Value Density Estimation for Imitation Learning and Goal-Conditioned Reinforcement Learning","date":"2020-02-15","arxiv_id":"2002.06473","repositories_listed":1,"syntology":null},{"url":"/paper/extended-markov-games-to-learn-multiple-tasks","slug":"extended-markov-games-to-learn-multiple-tasks","title":"Extended Markov Games to Learn Multiple Tasks in Multi-Agent Reinforcement Learning","date":"2020-02-14","arxiv_id":"2002.06000","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-via-adversarial-1","slug":"robust-reinforcement-learning-via-adversarial-1","title":"Robust Reinforcement Learning via Adversarial training with Langevin Dynamics","date":"2020-02-14","arxiv_id":"2002.06063","repositories_listed":1,"syntology":null},{"url":"/paper/effective-reinforcement-learning-through","slug":"effective-reinforcement-learning-through","title":"Effective Reinforcement Learning through Evolutionary Surrogate-Assisted Prescription","date":"2020-02-13","arxiv_id":"2002.05368","repositories_listed":1,"syntology":null},{"url":"/paper/hoplite-efficient-collective-communication","slug":"hoplite-efficient-collective-communication","title":"Hoplite: Efficient and Fault-Tolerant Collective Communication for Task-Based Distributed Systems","date":"2020-02-13","arxiv_id":"2002.05814","repositories_listed":1,"syntology":null},{"url":"/paper/provably-convergent-policy-gradient-methods","slug":"provably-convergent-policy-gradient-methods","title":"On the Convergence Theory of Debiased Model-Agnostic Meta-Reinforcement Learning","date":"2020-02-12","arxiv_id":"2002.05135","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-quantum","slug":"reinforcement-learning-enhanced-quantum","title":"Reinforcement Learning Enhanced Quantum-inspired Algorithm for Combinatorial Optimization","date":"2020-02-11","arxiv_id":"2002.04676","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-enhanced-quantum#ran","syntology_url":"https://syntology.ai/paper/2002.04676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.04676"}},"official":{"repos":["BeloborodovDS/SIMCIM-RL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/discrete-action-on-policy-learning-with","slug":"discrete-action-on-policy-learning-with","title":"Discrete Action On-Policy Learning with Action-Value Critic","date":"2020-02-10","arxiv_id":"2002.03534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-action-on-policy-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.03534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03534"}},"official":{"repos":["yuguangyue/CARSM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sparseids-learning-packet-sampling-with","slug":"sparseids-learning-packet-sampling-with","title":"SparseIDS: Learning Packet Sampling with Reinforcement Learning","date":"2020-02-10","arxiv_id":"2002.03872","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-portfolio","slug":"reinforcement-learning-based-portfolio","title":"Reinforcement-Learning based Portfolio Management with Augmented Asset Movement Prediction States","date":"2020-02-09","arxiv_id":"2002.05780","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/reinforcement-learning-based-portfolio#ran","syntology_url":"https://syntology.ai/paper/2002.05780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05780"}},"official":null}},{"url":"/paper/attractive-or-faithful-popularity-reinforced","slug":"attractive-or-faithful-popularity-reinforced","title":"Attractive or Faithful? Popularity-Reinforced Learning for Inspired Headline Generation","date":"2020-02-06","arxiv_id":"2002.02095","repositories_listed":1,"syntology":null},{"url":"/paper/multi-type-mean-field-reinforcement-learning","slug":"multi-type-mean-field-reinforcement-learning","title":"Multi Type Mean Field Reinforcement Learning","date":"2020-02-06","arxiv_id":"2002.02513","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-type-mean-field-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2002.02513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02513"}},"official":{"repos":["BorealisAI/mtmfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-time","slug":"a-reinforcement-learning-framework-for-time","title":"Dynamic Causal Effects Evaluation in A/B Testing with a Reinforcement Learning Framework","date":"2020-02-05","arxiv_id":"2002.01711","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-reinforcement-learning-framework-for-time#ran","syntology_url":"https://syntology.ai/paper/2002.01711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01711"}},"official":{"repos":["callmespring/causalrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-the-markov-decision-process-fit-the-data","slug":"does-the-markov-decision-process-fit-the-data","title":"Does the Markov Decision Process Fit the Data: Testing for the Markov Property in Sequential Decision Making","date":"2020-02-05","arxiv_id":"2002.01751","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/does-the-markov-decision-process-fit-the-data#ran","syntology_url":"https://syntology.ai/paper/2002.01751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01751"}},"official":null}},{"url":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/integrating-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.00434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00434"}},"official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/periodic-intra-ensemble-knowledge","slug":"periodic-intra-ensemble-knowledge","title":"Periodic Intra-Ensemble Knowledge Distillation for Reinforcement Learning","date":"2020-02-01","arxiv_id":"2002.00149","repositories_listed":1,"syntology":null},{"url":"/paper/improving-the-robustness-of-graphs-through","slug":"improving-the-robustness-of-graphs-through","title":"Goal-directed graph construction using reinforcement learning","date":"2020-01-30","arxiv_id":"2001.11279","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-calibration-of-coherent-state","slug":"real-time-calibration-of-coherent-state","title":"Real-time calibration of coherent-state receivers: learning by trial and error","date":"2020-01-28","arxiv_id":"2001.10283","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-and-countermeasures-for","slug":"challenges-and-countermeasures-for","title":"Challenges and Countermeasures for Adversarial Attacks on Deep Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09684","repositories_listed":1,"syntology":null},{"url":"/paper/computing-the-feedback-capacity-of-finite","slug":"computing-the-feedback-capacity-of-finite","title":"Computing the Feedback Capacity of Finite State Channels using Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09685","repositories_listed":1,"syntology":null},{"url":"/paper/rotation-translation-and-cropping-for-zero","slug":"rotation-translation-and-cropping-for-zero","title":"Rotation, Translation, and Cropping for Zero-Shot Generalization","date":"2020-01-27","arxiv_id":"2001.09908","repositories_listed":1,"syntology":null},{"url":"/paper/some-insights-into-lifelong-reinforcement","slug":"some-insights-into-lifelong-reinforcement","title":"Some Insights into Lifelong Reinforcement Learning Systems","date":"2020-01-27","arxiv_id":"2001.09608","repositories_listed":1,"syntology":null},{"url":"/paper/tractable-reinforcement-learning-of-signal","slug":"tractable-reinforcement-learning-of-signal","title":"Tractable Reinforcement Learning of Signal Temporal Logic Objectives","date":"2020-01-26","arxiv_id":"2001.09467","repositories_listed":1,"syntology":null},{"url":"/paper/graph-constrained-reinforcement-learning-for-1","slug":"graph-constrained-reinforcement-learning-for-1","title":"Graph Constrained Reinforcement Learning for Natural Language Action Spaces","date":"2020-01-23","arxiv_id":"2001.08837","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 5 unverified","sample_list":"/paper/graph-constrained-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2001.08837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08837"}},"official":{"repos":["rajammanabrolu/KG-A2C"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/glib-exploration-via-goal-literal-babbling","slug":"glib-exploration-via-goal-literal-babbling","title":"GLIB: Efficient Exploration for Relational Model-Based Reinforcement Learning via Goal-Literal Babbling","date":"2020-01-22","arxiv_id":"2001.08299","repositories_listed":1,"syntology":null},{"url":"/paper/on-simple-reactive-neural-networks-for","slug":"on-simple-reactive-neural-networks-for","title":"On Simple Reactive Neural Networks for Behaviour-Based Reinforcement Learning","date":"2020-01-22","arxiv_id":"2001.07973","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-pragmatics-from-referential-game","slug":"emergence-of-pragmatics-from-referential-game","title":"Emergence of Pragmatics from Referential Game between Theory of Mind Agents","date":"2020-01-21","arxiv_id":"2001.07752","repositories_listed":1,"syntology":null},{"url":"/paper/sarl-deep-reinforcement-learning-based-human","slug":"sarl-deep-reinforcement-learning-based-human","title":"SARL*: Deep Reinforcement Learning based Human-Aware Navigation for Mobile Robot in Indoor Environments","date":"2020-01-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/discriminator-soft-actor-critic-without","slug":"discriminator-soft-actor-critic-without","title":"Discriminator Soft Actor Critic without Extrinsic Rewards","date":"2020-01-19","arxiv_id":"2001.06808","repositories_listed":1,"syntology":null},{"url":"/paper/tree-structured-policy-based-progressive","slug":"tree-structured-policy-based-progressive","title":"Tree-Structured Policy based Progressive Reinforcement Learning for Temporally Language Grounding in Video","date":"2020-01-18","arxiv_id":"2001.06680","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tree-structured-policy-based-progressive#ran","syntology_url":"https://syntology.ai/paper/2001.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06680"}},"official":{"repos":["WuJie1010/TSP-PRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-action-reinforcement-learning-for","slug":"continuous-action-reinforcement-learning-for","title":"Continuous-action Reinforcement Learning for Playing Racing Games: Comparing SPG to PPO","date":"2020-01-15","arxiv_id":"2001.05270","repositories_listed":1,"syntology":null},{"url":"/paper/lipschitz-lifelong-reinforcement-learning-1","slug":"lipschitz-lifelong-reinforcement-learning-1","title":"Lipschitz Lifelong Reinforcement Learning","date":"2020-01-15","arxiv_id":"2001.05411","repositories_listed":1,"syntology":null},{"url":"/paper/pops-policy-pruning-and-shrinking-for-deep","slug":"pops-policy-pruning-and-shrinking-for-deep","title":"PoPS: Policy Pruning and Shrinking for Deep Reinforcement Learning","date":"2020-01-14","arxiv_id":"2001.05012","repositories_listed":1,"syntology":null},{"url":"/paper/popcorn-partially-observed-prediction","slug":"popcorn-partially-observed-prediction","title":"POPCORN: Partially Observed Prediction COnstrained ReiNforcement Learning","date":"2020-01-13","arxiv_id":"2001.04032","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/popcorn-partially-observed-prediction#ran","syntology_url":"https://syntology.ai/paper/2001.04032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04032"}},"official":{"repos":["dtak/POPCORN-POMDP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/statistical-inference-of-the-value-function","slug":"statistical-inference-of-the-value-function","title":"Statistical Inference of the Value Function for Reinforcement Learning in Infinite Horizon Settings","date":"2020-01-13","arxiv_id":"2001.04515","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/statistical-inference-of-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2001.04515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04515"}},"official":{"repos":["shengzhang37/SAVE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-engineering-for-object-pick-and-place","slug":"reward-engineering-for-object-pick-and-place","title":"Reward Engineering for Object Pick and Place Training","date":"2020-01-11","arxiv_id":"2001.03792","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-black-box-video-attack-with","slug":"sparse-black-box-video-attack-with","title":"Sparse Black-box Video Attack with Reinforcement Learning","date":"2020-01-11","arxiv_id":"2001.03754","repositories_listed":1,"syntology":null},{"url":"/paper/population-guided-parallel-policy-search-for-1","slug":"population-guided-parallel-policy-search-for-1","title":"Population-Guided Parallel Policy Search for Reinforcement Learning","date":"2020-01-09","arxiv_id":"2001.02907","repositories_listed":1,"syntology":null},{"url":"/paper/a-nonparametric-offpolicy-policy-gradient","slug":"a-nonparametric-offpolicy-policy-gradient","title":"A Nonparametric Off-Policy Policy Gradient","date":"2020-01-08","arxiv_id":"2001.02435","repositories_listed":1,"syntology":null},{"url":"/paper/blue-river-controls-a-toolkit-for","slug":"blue-river-controls-a-toolkit-for","title":"Blue River Controls: A toolkit for Reinforcement Learning Control Systems on Hardware","date":"2020-01-07","arxiv_id":"2001.02254","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-human","slug":"deep-reinforcement-learning-for-active-human","title":"Deep Reinforcement Learning for Active Human Pose Estimation","date":"2020-01-07","arxiv_id":"2001.02024","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-via-fenchel","slug":"reinforcement-learning-via-fenchel","title":"Reinforcement Learning via Fenchel-Rockafellar Duality","date":"2020-01-07","arxiv_id":"2001.01866","repositories_listed":1,"syntology":null},{"url":"/paper/a-boolean-task-algebra-for-reinforcement-1","slug":"a-boolean-task-algebra-for-reinforcement-1","title":"A Boolean Task Algebra for Reinforcement Learning","date":"2020-01-06","arxiv_id":"2001.01394","repositories_listed":1,"syntology":null},{"url":"/paper/an-optimistic-perspective-on-offline-deep","slug":"an-optimistic-perspective-on-offline-deep","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-f-gans-and-1","slug":"bridging-the-gap-between-f-gans-and-1","title":"Bridging the Gap Between f-GANs and Wasserstein GANs","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/curl-contrastive-unsupervised-representation","slug":"curl-contrastive-unsupervised-representation","title":"CURL: Contrastive Unsupervised Representation Learning for Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-in-synthetically","slug":"learning-to-navigate-in-synthetically","title":"Learning to Navigate in Synthetically Accessible Chemical Space Using Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/long-term-visitation-value-for-deep","slug":"long-term-visitation-value-for-deep","title":"Long-Term Visitation Value for Deep Exploration in Sparse Reward Reinforcement Learning","date":"2020-01-01","arxiv_id":"2001.00119","repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-with-autonomous-1","slug":"meta-reinforcement-learning-with-autonomous-1","title":"Meta Reinforcement Learning with Autonomous Inference of Subtask Dependencies","date":"2020-01-01","arxiv_id":"2001.00248","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-reinforcement-learning-with-autonomous-1#ran","syntology_url":"https://syntology.ai/paper/2001.00248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00248"}},"official":{"repos":["srsohn/msgi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-imitation-learning-with-diverse","slug":"variational-imitation-learning-with-diverse","title":"Variational Imitation Learning with Diverse-quality Demonstrations","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reward-conditioned-policies","slug":"reward-conditioned-policies","title":"Reward-Conditioned Policies","date":"2019-12-31","arxiv_id":"1912.13465","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-conditioned-policies#ran","syntology_url":"https://syntology.ai/paper/1912.13465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.13465"}},"official":null}},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/weak-supervision-for-fake-news-detection-via","slug":"weak-supervision-for-fake-news-detection-via","title":"Weak Supervision for Fake News Detection via Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12520","repositories_listed":1,"syntology":null},{"url":"/paper/discrete-and-continuous-action-representation","slug":"discrete-and-continuous-action-representation","title":"Discrete and Continuous Action Representation for Practical RL in Video Games","date":"2019-12-23","arxiv_id":"1912.11077","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-and-continuous-action-representation#ran","syntology_url":"https://syntology.ai/paper/1912.11077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11077"}},"official":null}},{"url":"/paper/learning-to-navigate-using-mid-level-visual","slug":"learning-to-navigate-using-mid-level-visual","title":"Learning to Navigate Using Mid-Level Visual Priors","date":"2019-12-23","arxiv_id":"1912.11121","repositories_listed":1,"syntology":null},{"url":"/paper/parameterized-indexed-value-function-for","slug":"parameterized-indexed-value-function-for","title":"Parameterized Indexed Value Function for Efficient Exploration in Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.10577","repositories_listed":1,"syntology":null},{"url":"/paper/towards-practical-multi-object-manipulation","slug":"towards-practical-multi-object-manipulation","title":"Towards Practical Multi-Object Manipulation using Relational Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.11032","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-practical-multi-object-manipulation#ran","syntology_url":"https://syntology.ai/paper/1912.11032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11032"}},"official":null}},{"url":"/paper/variational-recurrent-models-for-solving-1","slug":"variational-recurrent-models-for-solving-1","title":"Variational Recurrent Models for Solving Partially Observable Control Tasks","date":"2019-12-23","arxiv_id":"1912.10703","repositories_listed":1,"syntology":null},{"url":"/paper/can-agents-learn-by-analogy-an-inferable","slug":"can-agents-learn-by-analogy-an-inferable","title":"Can Agents Learn by Analogy? An Inferable Model for PAC Reinforcement Learning","date":"2019-12-21","arxiv_id":"1912.10329","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-reinforcement-learning-for","slug":"distributed-reinforcement-learning-for","title":"Distributed Reinforcement Learning for Decentralized Linear Quadratic Control: A Derivative-Free Policy Optimization Approach","date":"2019-12-19","arxiv_id":"1912.09135","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-for-1","slug":"distributional-reinforcement-learning-for-1","title":"Distributional Reinforcement Learning for Energy-Based Sequential Models","date":"2019-12-18","arxiv_id":"1912.08517","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/distributional-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/1912.08517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.08517"}},"official":{"repos":["parshakova/GAMS-for-Data-Efficient-Learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pixelrl-fully-convolutional-network-with","slug":"pixelrl-fully-convolutional-network-with","title":"PixelRL: Fully Convolutional Network with Reinforcement Learning for Image Processing","date":"2019-12-16","arxiv_id":"1912.07190","repositories_listed":1,"syntology":null},{"url":"/paper/unas-differentiable-architecture-search-meets","slug":"unas-differentiable-architecture-search-meets","title":"UNAS: Differentiable Architecture Search Meets Reinforcement Learning","date":"2019-12-16","arxiv_id":"1912.07651","repositories_listed":1,"syntology":null},{"url":"/paper/dota-2-with-large-scale-deep-reinforcement","slug":"dota-2-with-large-scale-deep-reinforcement","title":"Dota 2 with Large Scale Deep Reinforcement Learning","date":"2019-12-13","arxiv_id":"1912.06680","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dota-2-with-large-scale-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1912.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.06680"}},"official":null}},{"url":"/paper/the-playstation-reinforcement-learning","slug":"the-playstation-reinforcement-learning","title":"The PlayStation Reinforcement Learning Environment (PSXLE)","date":"2019-12-12","arxiv_id":"1912.06101","repositories_listed":1,"syntology":null},{"url":"/paper/smirl-surprise-minimizing-rl-in-dynamic","slug":"smirl-surprise-minimizing-rl-in-dynamic","title":"SMiRL: Surprise Minimizing Reinforcement Learning in Unstable Environments","date":"2019-12-11","arxiv_id":"1912.05510","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-reliability-of-reinforcement-1","slug":"measuring-the-reliability-of-reinforcement-1","title":"Measuring the Reliability of Reinforcement Learning Algorithms","date":"2019-12-10","arxiv_id":"1912.05663","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-the-reliability-of-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1912.05663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05663"}},"official":{"repos":["google-research/rl-reliability-metrics"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chainerrl-a-deep-reinforcement-learning","slug":"chainerrl-a-deep-reinforcement-learning","title":"ChainerRL: A Deep Reinforcement Learning Library","date":"2019-12-09","arxiv_id":"1912.03905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chainerrl-a-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1912.03905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03905"}},"official":{"repos":["chainer/chainerrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploratory-not-explanatory-counterfactual-1","slug":"exploratory-not-explanatory-counterfactual-1","title":"Exploratory Not Explanatory: Counterfactual Analysis of Saliency Maps for Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.05743","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-cooperative-multi-agent","slug":"hierarchical-cooperative-multi-agent","title":"Hierarchical Cooperative Multi-Agent Reinforcement Learning with Skill Discovery","date":"2019-12-07","arxiv_id":"1912.03558","repositories_listed":1,"syntology":null}],"record_sha256":"42e2d872bbfac99861fcc243762e1bbd8729e27612fbbfa0abcba6bf08774355","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}