{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/35","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":35,"pages_in_order":135,"rows_per_page":100,"rows":[3401,3500],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/34","next":"/task/reinforcement-learning-2/papers/36","papers":[{"url":"/paper/policy-teaching-via-environment-poisoning","slug":"policy-teaching-via-environment-poisoning","title":"Policy Teaching via Environment Poisoning: Training-time Adversarial Attacks against Reinforcement Learning","date":"2020-03-28","arxiv_id":"2003.12909","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-teaching-via-environment-poisoning#ran","syntology_url":"https://syntology.ai/paper/2003.12909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12909"}},"official":{"repos":["adishs/icml2020_rl-policy-teaching_code"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-learning-in-asset-management-part-2","slug":"machine-learning-in-asset-management-part-2","title":"Machine Learning in Asset Management—Part 2: Portfolio Construction—Weight Optimization. The Journal of Financial Data Science","date":"2020-03-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-investigation-of-the-challenges","slug":"an-empirical-investigation-of-the-challenges","title":"An empirical investigation of the challenges of real-world reinforcement learning","date":"2020-03-24","arxiv_id":"2003.11881","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-empirical-investigation-of-the-challenges#ran","syntology_url":"https://syntology.ai/paper/2003.11881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11881"}},"official":{"repos":["google-research/realworldrl_suite"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-population-curriculum-for-1","slug":"evolutionary-population-curriculum-for-1","title":"Evolutionary Population Curriculum for Scaling Multi-Agent Reinforcement Learning","date":"2020-03-23","arxiv_id":"2003.10423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolutionary-population-curriculum-for-1#ran","syntology_url":"https://syntology.ai/paper/2003.10423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10423"}},"official":{"repos":["qian18long/epciclr2020"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-deep-reinforcement-learning-methods-for","slug":"using-deep-reinforcement-learning-methods-for","title":"Using Deep Reinforcement Learning Methods for Autonomous Vessels in 2D Environments","date":"2020-03-23","arxiv_id":"2003.10249","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-of-control-affine","slug":"safe-reinforcement-learning-of-control-affine","title":"Safe Reinforcement Learning of Control-Affine Systems with Vertex Networks","date":"2020-03-20","arxiv_id":"2003.09488","repositories_listed":1,"syntology":null},{"url":"/paper/adjust-planning-strategies-to-accommodate","slug":"adjust-planning-strategies-to-accommodate","title":"Adjust Planning Strategies to Accommodate Reinforcement Learning Agents","date":"2020-03-19","arxiv_id":"2003.08554","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-fly-via-deep-model-based","slug":"learning-to-fly-via-deep-model-based","title":"Learning to Fly via Deep Model-Based Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08876","repositories_listed":1,"syntology":null},{"url":"/paper/monotonic-value-function-factorisation-for","slug":"monotonic-value-function-factorisation-for","title":"Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monotonic-value-function-factorisation-for#ran","syntology_url":"https://syntology.ai/paper/2003.08839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08839"}},"official":{"repos":["oxwhirl/pymarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/social-navigation-with-human-empowerment","slug":"social-navigation-with-human-empowerment","title":"Social Navigation with Human Empowerment driven Deep Reinforcement Learning","date":"2020-03-18","arxiv_id":"2003.08158","repositories_listed":1,"syntology":null},{"url":"/paper/giving-up-control-neurons-as-reinforcement","slug":"giving-up-control-neurons-as-reinforcement","title":"Giving Up Control: Neurons as Reinforcement Learning Agents","date":"2020-03-17","arxiv_id":"2003.11642","repositories_listed":1,"syntology":null},{"url":"/paper/simultaneous-navigation-and-radio-mapping-for","slug":"simultaneous-navigation-and-radio-mapping-for","title":"Simultaneous Navigation and Radio Mapping for Cellular-Connected UAV with Deep Reinforcement Learning","date":"2020-03-17","arxiv_id":"2003.07574","repositories_listed":1,"syntology":null},{"url":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-discovering-of-causal","slug":"self-supervised-discovering-of-causal","title":"Self-Supervised Discovering of Interpretable Features for Reinforcement Learning","date":"2020-03-16","arxiv_id":"2003.07069","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-discovering-of-causal#ran","syntology_url":"https://syntology.ai/paper/2003.07069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07069"}},"official":{"repos":["shiwj16/SSINet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-efficient-exploration-for-rl-with","slug":"provably-efficient-exploration-for-rl-with","title":"Provably Efficient Exploration for Reinforcement Learning Using Unsupervised Learning","date":"2020-03-15","arxiv_id":"2003.06898","repositories_listed":1,"syntology":null},{"url":"/paper/deep-deterministic-portfolio-optimization","slug":"deep-deterministic-portfolio-optimization","title":"Deep Deterministic Portfolio Optimization","date":"2020-03-13","arxiv_id":"2003.06497","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-deterministic-portfolio-optimization#ran","syntology_url":"https://syntology.ai/paper/2003.06497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06497"}},"official":{"repos":["CFMTech/Deep-RL-for-Portfolio-Optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/option-discovery-in-the-absence-of-rewards","slug":"option-discovery-in-the-absence-of-rewards","title":"Option Discovery in the Absence of Rewards with Manifold Analysis","date":"2020-03-12","arxiv_id":"2003.05878","repositories_listed":1,"syntology":null},{"url":"/paper/the-chefs-hat-simulation-environment-for","slug":"the-chefs-hat-simulation-environment-for","title":"The Chef's Hat Simulation Environment for Reinforcement-Learning-Based Agents","date":"2020-03-12","arxiv_id":"2003.05861","repositories_listed":1,"syntology":null},{"url":"/paper/explore-and-exploit-with-heterotic-line","slug":"explore-and-exploit-with-heterotic-line","title":"Explore and Exploit with Heterotic Line Bundle Models","date":"2020-03-10","arxiv_id":"2003.04817","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-robustness-of-cooperative-multi-agent","slug":"on-the-robustness-of-cooperative-multi-agent","title":"On the Robustness of Cooperative Multi-Agent Reinforcement Learning","date":"2020-03-08","arxiv_id":"2003.03722","repositories_listed":1,"syntology":null},{"url":"/paper/ig-rl-inductive-graph-reinforcement-learning","slug":"ig-rl-inductive-graph-reinforcement-learning","title":"IG-RL: Inductive Graph Reinforcement Learning for Massive-Scale Traffic Signal Control","date":"2020-03-06","arxiv_id":"2003.05738","repositories_listed":1,"syntology":null},{"url":"/paper/can-increasing-input-dimensionality-improve","slug":"can-increasing-input-dimensionality-improve","title":"Can Increasing Input Dimensionality Improve Deep Reinforcement Learning?","date":"2020-03-03","arxiv_id":"2003.01629","repositories_listed":1,"syntology":null},{"url":"/paper/contention-window-optimization-in-ieee","slug":"contention-window-optimization-in-ieee","title":"Contention Window Optimization in IEEE 802.11ax Networks with Deep Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01492","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-synaptic-plasticity-with-online","slug":"embodied-synaptic-plasticity-with-online","title":"Embodied Synaptic Plasticity with Online Reinforcement learning","date":"2020-03-03","arxiv_id":"2003.01431","repositories_listed":1,"syntology":null},{"url":"/paper/robust-market-making-via-adversarial","slug":"robust-market-making-via-adversarial","title":"Robust Market Making via Adversarial Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01820","repositories_listed":1,"syntology":null},{"url":"/paper/autophase-juggling-hls-phase-orderings-in","slug":"autophase-juggling-hls-phase-orderings-in","title":"AutoPhase: Juggling HLS Phase Orderings in Random Forests with Deep Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00671","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autophase-juggling-hls-phase-orderings-in#ran","syntology_url":"https://syntology.ai/paper/2003.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00671"}},"official":{"repos":["ucb-bar/autophase"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ppmc-training-algorithm-a-robot-independent","slug":"ppmc-training-algorithm-a-robot-independent","title":"PPMC RL Training Algorithm: Rough Terrain Intelligent Robots through Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.02655","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-stochastic-policy-gradient-algorithm","slug":"a-hybrid-stochastic-policy-gradient-algorithm","title":"A Hybrid Stochastic Policy Gradient Algorithm for Reinforcement Learning","date":"2020-03-01","arxiv_id":"2003.00430","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hybrid-stochastic-policy-gradient-algorithm#ran","syntology_url":"https://syntology.ai/paper/2003.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00430"}},"official":{"repos":["unc-optimization/ProxHSPGA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-catastrophic-interference-in-atari-2600","slug":"on-catastrophic-interference-in-atari-2600","title":"On Catastrophic Interference in Atari 2600 Games","date":"2020-02-28","arxiv_id":"2002.12499","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-robotic-nanofabrication-with","slug":"autonomous-robotic-nanofabrication-with","title":"Autonomous robotic nanofabrication with reinforcement learning","date":"2020-02-27","arxiv_id":"2002.11952","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-of-risk-constrained","slug":"reinforcement-learning-of-risk-constrained","title":"Reinforcement Learning of Risk-Constrained Policies in Markov Decision Processes","date":"2020-02-27","arxiv_id":"2002.12086","repositories_listed":1,"syntology":null},{"url":"/paper/training-adversarial-agents-to-exploit","slug":"training-adversarial-agents-to-exploit","title":"Training Adversarial Agents to Exploit Weaknesses in Deep Control Policies","date":"2020-02-27","arxiv_id":"2002.12078","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-control-for","slug":"efficient-reinforcement-learning-control-for","title":"Efficient reinforcement learning control for continuum robots based on Inexplicit Prior Knowledge","date":"2020-02-26","arxiv_id":"2002.11573","repositories_listed":1,"syntology":null},{"url":"/paper/mid-flight-propeller-failure-detection-and","slug":"mid-flight-propeller-failure-detection-and","title":"Mid-flight Propeller Failure Detection and Control of Propeller-deficient Quadcopter using Reinforcement Learning","date":"2020-02-26","arxiv_id":"2002.11564","repositories_listed":1,"syntology":null},{"url":"/paper/using-reinforcement-learning-in-the","slug":"using-reinforcement-learning-in-the","title":"Using Reinforcement Learning in the Algorithmic Trading Problem","date":"2020-02-26","arxiv_id":"2002.11523","repositories_listed":1,"syntology":null},{"url":"/paper/human-apprenticeship-learning-via-kernel","slug":"human-apprenticeship-learning-via-kernel","title":"Reward Shaping for Human Learning via Inverse Reinforcement Learning","date":"2020-02-25","arxiv_id":"2002.10904","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-deep-reinforcement-learning-with","slug":"off-policy-deep-reinforcement-learning-with","title":"Off-Policy Deep Reinforcement Learning with Analogous Disentangled Exploration","date":"2020-02-25","arxiv_id":"2002.10738","repositories_listed":1,"syntology":null},{"url":"/paper/whole-body-control-of-a-mobile-manipulator","slug":"whole-body-control-of-a-mobile-manipulator","title":"Whole-Body Control of a Mobile Manipulator using End-to-End Reinforcement Learning","date":"2020-02-25","arxiv_id":"2003.02637","repositories_listed":1,"syntology":null},{"url":"/paper/reconfigurable-intelligent-surface-assisted","slug":"reconfigurable-intelligent-surface-assisted","title":"Reconfigurable Intelligent Surface Assisted Multiuser MISO Systems Exploiting Deep Reinforcement Learning","date":"2020-02-24","arxiv_id":"2002.10072","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-for-probabilistic","slug":"safe-reinforcement-learning-for-probabilistic","title":"Safe reinforcement learning for probabilistic reachability and safety specifications: A Lyapunov-based approach","date":"2020-02-24","arxiv_id":"2002.10126","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-particle-filter-reinforcement-1","slug":"discriminative-particle-filter-reinforcement-1","title":"Discriminative Particle Filter Reinforcement Learning for Complex Partial Observations","date":"2020-02-23","arxiv_id":"2002.09884","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-framework-for-deep","slug":"reinforcement-learning-framework-for-deep","title":"Reinforcement Learning Framework for Deep Brain Stimulation Study","date":"2020-02-22","arxiv_id":"2002.10948","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-through","slug":"efficient-deep-reinforcement-learning-through","title":"Efficient Deep Reinforcement Learning via Adaptive Policy Transfer","date":"2020-02-19","arxiv_id":"2002.08037","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-avoid-being-eaten-by-a-grue","slug":"how-to-avoid-being-eaten-by-a-grue","title":"How To Avoid Being Eaten By a Grue: Exploration Strategies for Text-Adventure Agents","date":"2020-02-19","arxiv_id":"2002.08795","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-avoid-being-eaten-by-a-grue#ran","syntology_url":"https://syntology.ai/paper/2002.08795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08795"}},"official":{"repos":["rajammanabrolu/Q-BERT"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/sim2real-transfer-for-reinforcement-learning","slug":"sim2real-transfer-for-reinforcement-learning","title":"Sim2Real Transfer for Reinforcement Learning without Dynamics Randomization","date":"2020-02-19","arxiv_id":"2002.11635","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-estimator-selection-for-off-policy","slug":"adaptive-estimator-selection-for-off-policy","title":"Adaptive Estimator Selection for Off-Policy Evaluation","date":"2020-02-18","arxiv_id":"2002.07729","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-molecular-design","slug":"reinforcement-learning-for-molecular-design","title":"Reinforcement Learning for Molecular Design Guided by Quantum Mechanics","date":"2020-02-18","arxiv_id":"2002.07717","repositories_listed":1,"syntology":null},{"url":"/paper/control-frequency-adaptation-via-action","slug":"control-frequency-adaptation-via-action","title":"Control Frequency Adaptation via Action Persistence in Batch Reinforcement Learning","date":"2020-02-17","arxiv_id":"2002.06836","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/control-frequency-adaptation-via-action#ran","syntology_url":"https://syntology.ai/paper/2002.06836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06836"}},"official":{"repos":["albertometelli/pfqi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/r-maddpg-for-partially-observable","slug":"r-maddpg-for-partially-observable","title":"R-MADDPG for Partially Observable Environments and Limited Communication","date":"2020-02-16","arxiv_id":"2002.06684","repositories_listed":1,"syntology":null},{"url":"/paper/deep-rl-agent-for-a-real-time-action-strategy","slug":"deep-rl-agent-for-a-real-time-action-strategy","title":"Deep RL Agent for a Real-Time Action Strategy Game","date":"2020-02-15","arxiv_id":"2002.06290","repositories_listed":1,"syntology":null},{"url":"/paper/universal-value-density-estimation-for","slug":"universal-value-density-estimation-for","title":"Universal Value Density Estimation for Imitation Learning and Goal-Conditioned Reinforcement Learning","date":"2020-02-15","arxiv_id":"2002.06473","repositories_listed":1,"syntology":null},{"url":"/paper/extended-markov-games-to-learn-multiple-tasks","slug":"extended-markov-games-to-learn-multiple-tasks","title":"Extended Markov Games to Learn Multiple Tasks in Multi-Agent Reinforcement Learning","date":"2020-02-14","arxiv_id":"2002.06000","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-via-adversarial-1","slug":"robust-reinforcement-learning-via-adversarial-1","title":"Robust Reinforcement Learning via Adversarial training with Langevin Dynamics","date":"2020-02-14","arxiv_id":"2002.06063","repositories_listed":1,"syntology":null},{"url":"/paper/effective-reinforcement-learning-through","slug":"effective-reinforcement-learning-through","title":"Effective Reinforcement Learning through Evolutionary Surrogate-Assisted Prescription","date":"2020-02-13","arxiv_id":"2002.05368","repositories_listed":1,"syntology":null},{"url":"/paper/hoplite-efficient-collective-communication","slug":"hoplite-efficient-collective-communication","title":"Hoplite: Efficient and Fault-Tolerant Collective Communication for Task-Based Distributed Systems","date":"2020-02-13","arxiv_id":"2002.05814","repositories_listed":1,"syntology":null},{"url":"/paper/provably-convergent-policy-gradient-methods","slug":"provably-convergent-policy-gradient-methods","title":"On the Convergence Theory of Debiased Model-Agnostic Meta-Reinforcement Learning","date":"2020-02-12","arxiv_id":"2002.05135","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-quantum","slug":"reinforcement-learning-enhanced-quantum","title":"Reinforcement Learning Enhanced Quantum-inspired Algorithm for Combinatorial Optimization","date":"2020-02-11","arxiv_id":"2002.04676","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-enhanced-quantum#ran","syntology_url":"https://syntology.ai/paper/2002.04676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.04676"}},"official":{"repos":["BeloborodovDS/SIMCIM-RL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/sparseids-learning-packet-sampling-with","slug":"sparseids-learning-packet-sampling-with","title":"SparseIDS: Learning Packet Sampling with Reinforcement Learning","date":"2020-02-10","arxiv_id":"2002.03872","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-portfolio","slug":"reinforcement-learning-based-portfolio","title":"Reinforcement-Learning based Portfolio Management with Augmented Asset Movement Prediction States","date":"2020-02-09","arxiv_id":"2002.05780","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/reinforcement-learning-based-portfolio#ran","syntology_url":"https://syntology.ai/paper/2002.05780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05780"}},"official":null}},{"url":"/paper/multi-type-mean-field-reinforcement-learning","slug":"multi-type-mean-field-reinforcement-learning","title":"Multi Type Mean Field Reinforcement Learning","date":"2020-02-06","arxiv_id":"2002.02513","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-type-mean-field-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2002.02513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02513"}},"official":{"repos":["BorealisAI/mtmfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-time","slug":"a-reinforcement-learning-framework-for-time","title":"Dynamic Causal Effects Evaluation in A/B Testing with a Reinforcement Learning Framework","date":"2020-02-05","arxiv_id":"2002.01711","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-reinforcement-learning-framework-for-time#ran","syntology_url":"https://syntology.ai/paper/2002.01711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01711"}},"official":{"repos":["callmespring/causalrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-the-markov-decision-process-fit-the-data","slug":"does-the-markov-decision-process-fit-the-data","title":"Does the Markov Decision Process Fit the Data: Testing for the Markov Property in Sequential Decision Making","date":"2020-02-05","arxiv_id":"2002.01751","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/does-the-markov-decision-process-fit-the-data#ran","syntology_url":"https://syntology.ai/paper/2002.01751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01751"}},"official":null}},{"url":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/integrating-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.00434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00434"}},"official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/periodic-intra-ensemble-knowledge","slug":"periodic-intra-ensemble-knowledge","title":"Periodic Intra-Ensemble Knowledge Distillation for Reinforcement Learning","date":"2020-02-01","arxiv_id":"2002.00149","repositories_listed":1,"syntology":null},{"url":"/paper/improving-the-robustness-of-graphs-through","slug":"improving-the-robustness-of-graphs-through","title":"Goal-directed graph construction using reinforcement learning","date":"2020-01-30","arxiv_id":"2001.11279","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-and-countermeasures-for","slug":"challenges-and-countermeasures-for","title":"Challenges and Countermeasures for Adversarial Attacks on Deep Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09684","repositories_listed":1,"syntology":null},{"url":"/paper/computing-the-feedback-capacity-of-finite","slug":"computing-the-feedback-capacity-of-finite","title":"Computing the Feedback Capacity of Finite State Channels using Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09685","repositories_listed":1,"syntology":null},{"url":"/paper/some-insights-into-lifelong-reinforcement","slug":"some-insights-into-lifelong-reinforcement","title":"Some Insights into Lifelong Reinforcement Learning Systems","date":"2020-01-27","arxiv_id":"2001.09608","repositories_listed":1,"syntology":null},{"url":"/paper/tractable-reinforcement-learning-of-signal","slug":"tractable-reinforcement-learning-of-signal","title":"Tractable Reinforcement Learning of Signal Temporal Logic Objectives","date":"2020-01-26","arxiv_id":"2001.09467","repositories_listed":1,"syntology":null},{"url":"/paper/graph-constrained-reinforcement-learning-for-1","slug":"graph-constrained-reinforcement-learning-for-1","title":"Graph Constrained Reinforcement Learning for Natural Language Action Spaces","date":"2020-01-23","arxiv_id":"2001.08837","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 5 unverified","sample_list":"/paper/graph-constrained-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2001.08837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08837"}},"official":{"repos":["rajammanabrolu/KG-A2C"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/glib-exploration-via-goal-literal-babbling","slug":"glib-exploration-via-goal-literal-babbling","title":"GLIB: Efficient Exploration for Relational Model-Based Reinforcement Learning via Goal-Literal Babbling","date":"2020-01-22","arxiv_id":"2001.08299","repositories_listed":1,"syntology":null},{"url":"/paper/on-simple-reactive-neural-networks-for","slug":"on-simple-reactive-neural-networks-for","title":"On Simple Reactive Neural Networks for Behaviour-Based Reinforcement Learning","date":"2020-01-22","arxiv_id":"2001.07973","repositories_listed":1,"syntology":null},{"url":"/paper/sarl-deep-reinforcement-learning-based-human","slug":"sarl-deep-reinforcement-learning-based-human","title":"SARL*: Deep Reinforcement Learning based Human-Aware Navigation for Mobile Robot in Indoor Environments","date":"2020-01-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/discriminator-soft-actor-critic-without","slug":"discriminator-soft-actor-critic-without","title":"Discriminator Soft Actor Critic without Extrinsic Rewards","date":"2020-01-19","arxiv_id":"2001.06808","repositories_listed":1,"syntology":null},{"url":"/paper/tree-structured-policy-based-progressive","slug":"tree-structured-policy-based-progressive","title":"Tree-Structured Policy based Progressive Reinforcement Learning for Temporally Language Grounding in Video","date":"2020-01-18","arxiv_id":"2001.06680","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tree-structured-policy-based-progressive#ran","syntology_url":"https://syntology.ai/paper/2001.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06680"}},"official":{"repos":["WuJie1010/TSP-PRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-action-reinforcement-learning-for","slug":"continuous-action-reinforcement-learning-for","title":"Continuous-action Reinforcement Learning for Playing Racing Games: Comparing SPG to PPO","date":"2020-01-15","arxiv_id":"2001.05270","repositories_listed":1,"syntology":null},{"url":"/paper/lipschitz-lifelong-reinforcement-learning-1","slug":"lipschitz-lifelong-reinforcement-learning-1","title":"Lipschitz Lifelong Reinforcement Learning","date":"2020-01-15","arxiv_id":"2001.05411","repositories_listed":1,"syntology":null},{"url":"/paper/pops-policy-pruning-and-shrinking-for-deep","slug":"pops-policy-pruning-and-shrinking-for-deep","title":"PoPS: Policy Pruning and Shrinking for Deep Reinforcement Learning","date":"2020-01-14","arxiv_id":"2001.05012","repositories_listed":1,"syntology":null},{"url":"/paper/popcorn-partially-observed-prediction","slug":"popcorn-partially-observed-prediction","title":"POPCORN: Partially Observed Prediction COnstrained ReiNforcement Learning","date":"2020-01-13","arxiv_id":"2001.04032","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/popcorn-partially-observed-prediction#ran","syntology_url":"https://syntology.ai/paper/2001.04032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04032"}},"official":{"repos":["dtak/POPCORN-POMDP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/statistical-inference-of-the-value-function","slug":"statistical-inference-of-the-value-function","title":"Statistical Inference of the Value Function for Reinforcement Learning in Infinite Horizon Settings","date":"2020-01-13","arxiv_id":"2001.04515","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/statistical-inference-of-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2001.04515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04515"}},"official":{"repos":["shengzhang37/SAVE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-engineering-for-object-pick-and-place","slug":"reward-engineering-for-object-pick-and-place","title":"Reward Engineering for Object Pick and Place Training","date":"2020-01-11","arxiv_id":"2001.03792","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-black-box-video-attack-with","slug":"sparse-black-box-video-attack-with","title":"Sparse Black-box Video Attack with Reinforcement Learning","date":"2020-01-11","arxiv_id":"2001.03754","repositories_listed":1,"syntology":null},{"url":"/paper/population-guided-parallel-policy-search-for-1","slug":"population-guided-parallel-policy-search-for-1","title":"Population-Guided Parallel Policy Search for Reinforcement Learning","date":"2020-01-09","arxiv_id":"2001.02907","repositories_listed":1,"syntology":null},{"url":"/paper/blue-river-controls-a-toolkit-for","slug":"blue-river-controls-a-toolkit-for","title":"Blue River Controls: A toolkit for Reinforcement Learning Control Systems on Hardware","date":"2020-01-07","arxiv_id":"2001.02254","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-human","slug":"deep-reinforcement-learning-for-active-human","title":"Deep Reinforcement Learning for Active Human Pose Estimation","date":"2020-01-07","arxiv_id":"2001.02024","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-via-fenchel","slug":"reinforcement-learning-via-fenchel","title":"Reinforcement Learning via Fenchel-Rockafellar Duality","date":"2020-01-07","arxiv_id":"2001.01866","repositories_listed":1,"syntology":null},{"url":"/paper/a-boolean-task-algebra-for-reinforcement-1","slug":"a-boolean-task-algebra-for-reinforcement-1","title":"A Boolean Task Algebra for Reinforcement Learning","date":"2020-01-06","arxiv_id":"2001.01394","repositories_listed":1,"syntology":null},{"url":"/paper/an-optimistic-perspective-on-offline-deep","slug":"an-optimistic-perspective-on-offline-deep","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/curl-contrastive-unsupervised-representation","slug":"curl-contrastive-unsupervised-representation","title":"CURL: Contrastive Unsupervised Representation Learning for Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-in-synthetically","slug":"learning-to-navigate-in-synthetically","title":"Learning to Navigate in Synthetically Accessible Chemical Space Using Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/long-term-visitation-value-for-deep","slug":"long-term-visitation-value-for-deep","title":"Long-Term Visitation Value for Deep Exploration in Sparse Reward Reinforcement Learning","date":"2020-01-01","arxiv_id":"2001.00119","repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-with-autonomous-1","slug":"meta-reinforcement-learning-with-autonomous-1","title":"Meta Reinforcement Learning with Autonomous Inference of Subtask Dependencies","date":"2020-01-01","arxiv_id":"2001.00248","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-reinforcement-learning-with-autonomous-1#ran","syntology_url":"https://syntology.ai/paper/2001.00248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00248"}},"official":{"repos":["srsohn/msgi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-imitation-learning-with-diverse","slug":"variational-imitation-learning-with-diverse","title":"Variational Imitation Learning with Diverse-quality Demonstrations","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reward-conditioned-policies","slug":"reward-conditioned-policies","title":"Reward-Conditioned Policies","date":"2019-12-31","arxiv_id":"1912.13465","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-conditioned-policies#ran","syntology_url":"https://syntology.ai/paper/1912.13465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.13465"}},"official":null}},{"url":"/paper/weak-supervision-for-fake-news-detection-via","slug":"weak-supervision-for-fake-news-detection-via","title":"Weak Supervision for Fake News Detection via Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12520","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-using-mid-level-visual","slug":"learning-to-navigate-using-mid-level-visual","title":"Learning to Navigate Using Mid-Level Visual Priors","date":"2019-12-23","arxiv_id":"1912.11121","repositories_listed":1,"syntology":null},{"url":"/paper/parameterized-indexed-value-function-for","slug":"parameterized-indexed-value-function-for","title":"Parameterized Indexed Value Function for Efficient Exploration in Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.10577","repositories_listed":1,"syntology":null},{"url":"/paper/towards-practical-multi-object-manipulation","slug":"towards-practical-multi-object-manipulation","title":"Towards Practical Multi-Object Manipulation using Relational Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.11032","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-practical-multi-object-manipulation#ran","syntology_url":"https://syntology.ai/paper/1912.11032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11032"}},"official":null}},{"url":"/paper/can-agents-learn-by-analogy-an-inferable","slug":"can-agents-learn-by-analogy-an-inferable","title":"Can Agents Learn by Analogy? An Inferable Model for PAC Reinforcement Learning","date":"2019-12-21","arxiv_id":"1912.10329","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-for-1","slug":"distributional-reinforcement-learning-for-1","title":"Distributional Reinforcement Learning for Energy-Based Sequential Models","date":"2019-12-18","arxiv_id":"1912.08517","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/distributional-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/1912.08517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.08517"}},"official":{"repos":["parshakova/GAMS-for-Data-Efficient-Learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"8db12bd4eb10405aa6cc6284713a31b5904acd1b8b00890b42455931976ad712","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}