{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/32","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":32,"pages_in_order":132,"rows_per_page":100,"rows":[3101,3200],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/31","next":"/task/reinforcement-learning/papers/33","papers":[{"url":"/paper/explore-discover-and-learn-unsupervised","slug":"explore-discover-and-learn-unsupervised","title":"Explore, Discover and Learn: Unsupervised Discovery of State-Covering Skills","date":"2020-02-10","arxiv_id":"2002.03647","repositories_listed":1,"syntology":null},{"url":"/paper/self-assttentive-associative-memory","slug":"self-assttentive-associative-memory","title":"Self-Attentive Associative Memory","date":"2020-02-10","arxiv_id":"2002.03519","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-assttentive-associative-memory#ran","syntology_url":"https://syntology.ai/paper/2002.03519","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03519"}},"official":{"repos":["thaihungle/SAM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sparseids-learning-packet-sampling-with","slug":"sparseids-learning-packet-sampling-with","title":"SparseIDS: Learning Packet Sampling with Reinforcement Learning","date":"2020-02-10","arxiv_id":"2002.03872","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-portfolio","slug":"reinforcement-learning-based-portfolio","title":"Reinforcement-Learning based Portfolio Management with Augmented Asset Movement Prediction States","date":"2020-02-09","arxiv_id":"2002.05780","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/reinforcement-learning-based-portfolio#ran","syntology_url":"https://syntology.ai/paper/2002.05780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05780"}},"official":null}},{"url":"/paper/provably-efficient-adaptive-approximate","slug":"provably-efficient-adaptive-approximate","title":"Adaptive Approximate Policy Iteration","date":"2020-02-08","arxiv_id":"2002.03069","repositories_listed":1,"syntology":null},{"url":"/paper/an-initial-investigation-on-optimizing-tandem","slug":"an-initial-investigation-on-optimizing-tandem","title":"An initial investigation on optimizing tandem speaker verification and countermeasure systems using reinforcement learning","date":"2020-02-06","arxiv_id":"2002.03801","repositories_listed":1,"syntology":null},{"url":"/paper/attractive-or-faithful-popularity-reinforced","slug":"attractive-or-faithful-popularity-reinforced","title":"Attractive or Faithful? Popularity-Reinforced Learning for Inspired Headline Generation","date":"2020-02-06","arxiv_id":"2002.02095","repositories_listed":1,"syntology":null},{"url":"/paper/multi-type-mean-field-reinforcement-learning","slug":"multi-type-mean-field-reinforcement-learning","title":"Multi Type Mean Field Reinforcement Learning","date":"2020-02-06","arxiv_id":"2002.02513","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-type-mean-field-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2002.02513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02513"}},"official":{"repos":["BorealisAI/mtmfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-time","slug":"a-reinforcement-learning-framework-for-time","title":"Dynamic Causal Effects Evaluation in A/B Testing with a Reinforcement Learning Framework","date":"2020-02-05","arxiv_id":"2002.01711","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-reinforcement-learning-framework-for-time#ran","syntology_url":"https://syntology.ai/paper/2002.01711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01711"}},"official":{"repos":["callmespring/causalrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-the-markov-decision-process-fit-the-data","slug":"does-the-markov-decision-process-fit-the-data","title":"Does the Markov Decision Process Fit the Data: Testing for the Markov Property in Sequential Decision Making","date":"2020-02-05","arxiv_id":"2002.01751","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/does-the-markov-decision-process-fit-the-data#ran","syntology_url":"https://syntology.ai/paper/2002.01751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01751"}},"official":null}},{"url":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/integrating-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.00434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00434"}},"official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/periodic-intra-ensemble-knowledge","slug":"periodic-intra-ensemble-knowledge","title":"Periodic Intra-Ensemble Knowledge Distillation for Reinforcement Learning","date":"2020-02-01","arxiv_id":"2002.00149","repositories_listed":1,"syntology":null},{"url":"/paper/improving-the-robustness-of-graphs-through","slug":"improving-the-robustness-of-graphs-through","title":"Goal-directed graph construction using reinforcement learning","date":"2020-01-30","arxiv_id":"2001.11279","repositories_listed":1,"syntology":null},{"url":"/paper/gradientdice-rethinking-generalized-offline","slug":"gradientdice-rethinking-generalized-offline","title":"GradientDICE: Rethinking Generalized Offline Estimation of Stationary Values","date":"2020-01-29","arxiv_id":"2001.11113","repositories_listed":1,"syntology":null},{"url":"/paper/dont-feed-the-troll-detecting-troll-behavior","slug":"dont-feed-the-troll-detecting-troll-behavior","title":"Detecting Troll Behavior via Inverse Reinforcement Learning: A Case Study of Russian Trolls in the 2016 US Election","date":"2020-01-28","arxiv_id":"2001.10570","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-calibration-of-coherent-state","slug":"real-time-calibration-of-coherent-state","title":"Real-time calibration of coherent-state receivers: learning by trial and error","date":"2020-01-28","arxiv_id":"2001.10283","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-and-countermeasures-for","slug":"challenges-and-countermeasures-for","title":"Challenges and Countermeasures for Adversarial Attacks on Deep Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09684","repositories_listed":1,"syntology":null},{"url":"/paper/computing-the-feedback-capacity-of-finite","slug":"computing-the-feedback-capacity-of-finite","title":"Computing the Feedback Capacity of Finite State Channels using Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09685","repositories_listed":1,"syntology":null},{"url":"/paper/rotation-translation-and-cropping-for-zero","slug":"rotation-translation-and-cropping-for-zero","title":"Rotation, Translation, and Cropping for Zero-Shot Generalization","date":"2020-01-27","arxiv_id":"2001.09908","repositories_listed":1,"syntology":null},{"url":"/paper/some-insights-into-lifelong-reinforcement","slug":"some-insights-into-lifelong-reinforcement","title":"Some Insights into Lifelong Reinforcement Learning Systems","date":"2020-01-27","arxiv_id":"2001.09608","repositories_listed":1,"syntology":null},{"url":"/paper/graphaf-a-flow-based-autoregressive-model-for-1","slug":"graphaf-a-flow-based-autoregressive-model-for-1","title":"GraphAF: a Flow-based Autoregressive Model for Molecular Graph Generation","date":"2020-01-26","arxiv_id":"2001.09382","repositories_listed":1,"syntology":null},{"url":"/paper/tractable-reinforcement-learning-of-signal","slug":"tractable-reinforcement-learning-of-signal","title":"Tractable Reinforcement Learning of Signal Temporal Logic Objectives","date":"2020-01-26","arxiv_id":"2001.09467","repositories_listed":1,"syntology":null},{"url":"/paper/graph-constrained-reinforcement-learning-for-1","slug":"graph-constrained-reinforcement-learning-for-1","title":"Graph Constrained Reinforcement Learning for Natural Language Action Spaces","date":"2020-01-23","arxiv_id":"2001.08837","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 5 unverified","sample_list":"/paper/graph-constrained-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2001.08837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08837"}},"official":{"repos":["rajammanabrolu/KG-A2C"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/on-simple-reactive-neural-networks-for","slug":"on-simple-reactive-neural-networks-for","title":"On Simple Reactive Neural Networks for Behaviour-Based Reinforcement Learning","date":"2020-01-22","arxiv_id":"2001.07973","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-pragmatics-from-referential-game","slug":"emergence-of-pragmatics-from-referential-game","title":"Emergence of Pragmatics from Referential Game between Theory of Mind Agents","date":"2020-01-21","arxiv_id":"2001.07752","repositories_listed":1,"syntology":null},{"url":"/paper/sarl-deep-reinforcement-learning-based-human","slug":"sarl-deep-reinforcement-learning-based-human","title":"SARL*: Deep Reinforcement Learning based Human-Aware Navigation for Mobile Robot in Indoor Environments","date":"2020-01-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/discriminator-soft-actor-critic-without","slug":"discriminator-soft-actor-critic-without","title":"Discriminator Soft Actor Critic without Extrinsic Rewards","date":"2020-01-19","arxiv_id":"2001.06808","repositories_listed":1,"syntology":null},{"url":"/paper/tree-structured-policy-based-progressive","slug":"tree-structured-policy-based-progressive","title":"Tree-Structured Policy based Progressive Reinforcement Learning for Temporally Language Grounding in Video","date":"2020-01-18","arxiv_id":"2001.06680","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tree-structured-policy-based-progressive#ran","syntology_url":"https://syntology.ai/paper/2001.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06680"}},"official":{"repos":["WuJie1010/TSP-PRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-action-reinforcement-learning-for","slug":"continuous-action-reinforcement-learning-for","title":"Continuous-action Reinforcement Learning for Playing Racing Games: Comparing SPG to PPO","date":"2020-01-15","arxiv_id":"2001.05270","repositories_listed":1,"syntology":null},{"url":"/paper/lipschitz-lifelong-reinforcement-learning-1","slug":"lipschitz-lifelong-reinforcement-learning-1","title":"Lipschitz Lifelong Reinforcement Learning","date":"2020-01-15","arxiv_id":"2001.05411","repositories_listed":1,"syntology":null},{"url":"/paper/pops-policy-pruning-and-shrinking-for-deep","slug":"pops-policy-pruning-and-shrinking-for-deep","title":"PoPS: Policy Pruning and Shrinking for Deep Reinforcement Learning","date":"2020-01-14","arxiv_id":"2001.05012","repositories_listed":1,"syntology":null},{"url":"/paper/popcorn-partially-observed-prediction","slug":"popcorn-partially-observed-prediction","title":"POPCORN: Partially Observed Prediction COnstrained ReiNforcement Learning","date":"2020-01-13","arxiv_id":"2001.04032","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/popcorn-partially-observed-prediction#ran","syntology_url":"https://syntology.ai/paper/2001.04032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04032"}},"official":{"repos":["dtak/POPCORN-POMDP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/statistical-inference-of-the-value-function","slug":"statistical-inference-of-the-value-function","title":"Statistical Inference of the Value Function for Reinforcement Learning in Infinite Horizon Settings","date":"2020-01-13","arxiv_id":"2001.04515","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/statistical-inference-of-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2001.04515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04515"}},"official":{"repos":["shengzhang37/SAVE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-engineering-for-object-pick-and-place","slug":"reward-engineering-for-object-pick-and-place","title":"Reward Engineering for Object Pick and Place Training","date":"2020-01-11","arxiv_id":"2001.03792","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-black-box-video-attack-with","slug":"sparse-black-box-video-attack-with","title":"Sparse Black-box Video Attack with Reinforcement Learning","date":"2020-01-11","arxiv_id":"2001.03754","repositories_listed":1,"syntology":null},{"url":"/paper/closed-loop-deep-learning-generating-forward","slug":"closed-loop-deep-learning-generating-forward","title":"Closed-loop deep learning: generating forward models with back-propagation","date":"2020-01-09","arxiv_id":"2001.02970","repositories_listed":1,"syntology":null},{"url":"/paper/population-guided-parallel-policy-search-for-1","slug":"population-guided-parallel-policy-search-for-1","title":"Population-Guided Parallel Policy Search for Reinforcement Learning","date":"2020-01-09","arxiv_id":"2001.02907","repositories_listed":1,"syntology":null},{"url":"/paper/a-nonparametric-offpolicy-policy-gradient","slug":"a-nonparametric-offpolicy-policy-gradient","title":"A Nonparametric Off-Policy Policy Gradient","date":"2020-01-08","arxiv_id":"2001.02435","repositories_listed":1,"syntology":null},{"url":"/paper/blue-river-controls-a-toolkit-for","slug":"blue-river-controls-a-toolkit-for","title":"Blue River Controls: A toolkit for Reinforcement Learning Control Systems on Hardware","date":"2020-01-07","arxiv_id":"2001.02254","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-human","slug":"deep-reinforcement-learning-for-active-human","title":"Deep Reinforcement Learning for Active Human Pose Estimation","date":"2020-01-07","arxiv_id":"2001.02024","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-via-fenchel","slug":"reinforcement-learning-via-fenchel","title":"Reinforcement Learning via Fenchel-Rockafellar Duality","date":"2020-01-07","arxiv_id":"2001.01866","repositories_listed":1,"syntology":null},{"url":"/paper/a-boolean-task-algebra-for-reinforcement-1","slug":"a-boolean-task-algebra-for-reinforcement-1","title":"A Boolean Task Algebra for Reinforcement Learning","date":"2020-01-06","arxiv_id":"2001.01394","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-forecasts-in-unknown-environments","slug":"trajectory-forecasts-in-unknown-environments","title":"Trajectory Forecasts in Unknown Environments Conditioned on Grid-Based Plans","date":"2020-01-03","arxiv_id":"2001.00735","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trajectory-forecasts-in-unknown-environments#ran","syntology_url":"https://syntology.ai/paper/2001.00735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00735"}},"official":{"repos":["nachiket92/P2T"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-optimistic-perspective-on-offline-deep","slug":"an-optimistic-perspective-on-offline-deep","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/curl-contrastive-unsupervised-representation","slug":"curl-contrastive-unsupervised-representation","title":"CURL: Contrastive Unsupervised Representation Learning for Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/long-term-visitation-value-for-deep","slug":"long-term-visitation-value-for-deep","title":"Long-Term Visitation Value for Deep Exploration in Sparse Reward Reinforcement Learning","date":"2020-01-01","arxiv_id":"2001.00119","repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-with-autonomous-1","slug":"meta-reinforcement-learning-with-autonomous-1","title":"Meta Reinforcement Learning with Autonomous Inference of Subtask Dependencies","date":"2020-01-01","arxiv_id":"2001.00248","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-reinforcement-learning-with-autonomous-1#ran","syntology_url":"https://syntology.ai/paper/2001.00248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00248"}},"official":{"repos":["srsohn/msgi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-correlated-monte-carlo-for-1","slug":"adaptive-correlated-monte-carlo-for-1","title":"Adaptive Correlated Monte Carlo for Contextual Categorical Sequence Generation","date":"2019-12-31","arxiv_id":"1912.13151","repositories_listed":1,"syntology":null},{"url":"/paper/reward-conditioned-policies","slug":"reward-conditioned-policies","title":"Reward-Conditioned Policies","date":"2019-12-31","arxiv_id":"1912.13465","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-conditioned-policies#ran","syntology_url":"https://syntology.ai/paper/1912.13465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.13465"}},"official":null}},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/weak-supervision-for-fake-news-detection-via","slug":"weak-supervision-for-fake-news-detection-via","title":"Weak Supervision for Fake News Detection via Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12520","repositories_listed":1,"syntology":null},{"url":"/paper/discrete-and-continuous-action-representation","slug":"discrete-and-continuous-action-representation","title":"Discrete and Continuous Action Representation for Practical RL in Video Games","date":"2019-12-23","arxiv_id":"1912.11077","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-and-continuous-action-representation#ran","syntology_url":"https://syntology.ai/paper/1912.11077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11077"}},"official":null}},{"url":"/paper/learning-an-interpretable-traffic-signal","slug":"learning-an-interpretable-traffic-signal","title":"Learning an Interpretable Traffic Signal Control Policy","date":"2019-12-23","arxiv_id":"1912.11023","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-using-mid-level-visual","slug":"learning-to-navigate-using-mid-level-visual","title":"Learning to Navigate Using Mid-Level Visual Priors","date":"2019-12-23","arxiv_id":"1912.11121","repositories_listed":1,"syntology":null},{"url":"/paper/learning-variable-ordering-heuristics-for","slug":"learning-variable-ordering-heuristics-for","title":"Learning Variable Ordering Heuristics for Solving Constraint Satisfaction Problems","date":"2019-12-23","arxiv_id":"1912.10762","repositories_listed":1,"syntology":null},{"url":"/paper/parameterized-indexed-value-function-for","slug":"parameterized-indexed-value-function-for","title":"Parameterized Indexed Value Function for Efficient Exploration in Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.10577","repositories_listed":1,"syntology":null},{"url":"/paper/towards-practical-multi-object-manipulation","slug":"towards-practical-multi-object-manipulation","title":"Towards Practical Multi-Object Manipulation using Relational Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.11032","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-practical-multi-object-manipulation#ran","syntology_url":"https://syntology.ai/paper/1912.11032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11032"}},"official":null}},{"url":"/paper/variational-recurrent-models-for-solving-1","slug":"variational-recurrent-models-for-solving-1","title":"Variational Recurrent Models for Solving Partially Observable Control Tasks","date":"2019-12-23","arxiv_id":"1912.10703","repositories_listed":1,"syntology":null},{"url":"/paper/can-agents-learn-by-analogy-an-inferable","slug":"can-agents-learn-by-analogy-an-inferable","title":"Can Agents Learn by Analogy? An Inferable Model for PAC Reinforcement Learning","date":"2019-12-21","arxiv_id":"1912.10329","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-reinforcement-learning-for","slug":"distributed-reinforcement-learning-for","title":"Distributed Reinforcement Learning for Decentralized Linear Quadratic Control: A Derivative-Free Policy Optimization Approach","date":"2019-12-19","arxiv_id":"1912.09135","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-sensitive-learning-and-planning-1","slug":"uncertainty-sensitive-learning-and-planning-1","title":"Uncertainty-sensitive Learning and Planning with Ensembles","date":"2019-12-19","arxiv_id":"1912.09996","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reinforcement-learning-for-1","slug":"distributional-reinforcement-learning-for-1","title":"Distributional Reinforcement Learning for Energy-Based Sequential Models","date":"2019-12-18","arxiv_id":"1912.08517","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/distributional-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/1912.08517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.08517"}},"official":{"repos":["parshakova/GAMS-for-Data-Efficient-Learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pixelrl-fully-convolutional-network-with","slug":"pixelrl-fully-convolutional-network-with","title":"PixelRL: Fully Convolutional Network with Reinforcement Learning for Image Processing","date":"2019-12-16","arxiv_id":"1912.07190","repositories_listed":1,"syntology":null},{"url":"/paper/unas-differentiable-architecture-search-meets","slug":"unas-differentiable-architecture-search-meets","title":"UNAS: Differentiable Architecture Search Meets Reinforcement Learning","date":"2019-12-16","arxiv_id":"1912.07651","repositories_listed":1,"syntology":null},{"url":"/paper/dota-2-with-large-scale-deep-reinforcement","slug":"dota-2-with-large-scale-deep-reinforcement","title":"Dota 2 with Large Scale Deep Reinforcement Learning","date":"2019-12-13","arxiv_id":"1912.06680","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dota-2-with-large-scale-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1912.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.06680"}},"official":null}},{"url":"/paper/learning-improvement-heuristics-for-solving","slug":"learning-improvement-heuristics-for-solving","title":"Learning Improvement Heuristics for Solving Routing Problems","date":"2019-12-12","arxiv_id":"1912.05784","repositories_listed":1,"syntology":null},{"url":"/paper/the-playstation-reinforcement-learning","slug":"the-playstation-reinforcement-learning","title":"The PlayStation Reinforcement Learning Environment (PSXLE)","date":"2019-12-12","arxiv_id":"1912.06101","repositories_listed":1,"syntology":null},{"url":"/paper/efficacy-of-modern-neuro-evolutionary","slug":"efficacy-of-modern-neuro-evolutionary","title":"Efficacy of Modern Neuro-Evolutionary Strategies for Continuous Control Optimization","date":"2019-12-11","arxiv_id":"1912.05239","repositories_listed":1,"syntology":null},{"url":"/paper/smirl-surprise-minimizing-rl-in-dynamic","slug":"smirl-surprise-minimizing-rl-in-dynamic","title":"SMiRL: Surprise Minimizing Reinforcement Learning in Unstable Environments","date":"2019-12-11","arxiv_id":"1912.05510","repositories_listed":1,"syntology":null},{"url":"/paper/uct-adp-progressive-bias-algorithm-for","slug":"uct-adp-progressive-bias-algorithm-for","title":"UCT-ADP Progressive Bias Algorithm for Solving Gomoku","date":"2019-12-11","arxiv_id":"1912.05407","repositories_listed":1,"syntology":null},{"url":"/paper/deep-symbolic-regression-recovering","slug":"deep-symbolic-regression-recovering","title":"Deep symbolic regression: Recovering mathematical expressions from data via risk-seeking policy gradients","date":"2019-12-10","arxiv_id":"1912.04871","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-reliability-of-reinforcement-1","slug":"measuring-the-reliability-of-reinforcement-1","title":"Measuring the Reliability of Reinforcement Learning Algorithms","date":"2019-12-10","arxiv_id":"1912.05663","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-the-reliability-of-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1912.05663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05663"}},"official":{"repos":["google-research/rl-reliability-metrics"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chainerrl-a-deep-reinforcement-learning","slug":"chainerrl-a-deep-reinforcement-learning","title":"ChainerRL: A Deep Reinforcement Learning Library","date":"2019-12-09","arxiv_id":"1912.03905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chainerrl-a-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1912.03905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03905"}},"official":{"repos":["chainer/chainerrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploratory-not-explanatory-counterfactual-1","slug":"exploratory-not-explanatory-counterfactual-1","title":"Exploratory Not Explanatory: Counterfactual Analysis of Saliency Maps for Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.05743","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-cooperative-multi-agent","slug":"hierarchical-cooperative-multi-agent","title":"Hierarchical Cooperative Multi-Agent Reinforcement Learning with Skill Discovery","date":"2019-12-07","arxiv_id":"1912.03558","repositories_listed":1,"syntology":null},{"url":"/paper/valan-vision-and-language-agent-navigation","slug":"valan-vision-and-language-agent-navigation","title":"VALAN: Vision and Language Agent Navigation","date":"2019-12-06","arxiv_id":"1912.03241","repositories_listed":1,"syntology":null},{"url":"/paper/191202368","slug":"191202368","title":"Inter-Level Cooperation in Hierarchical Reinforcement Learning","date":"2019-12-05","arxiv_id":"1912.02368","repositories_listed":1,"syntology":null},{"url":"/paper/blind-inpainting-of-large-scale-masks-of-thin","slug":"blind-inpainting-of-large-scale-masks-of-thin","title":"Blind Inpainting of Large-scale Masks of Thin Structures with Adversarial and Reinforcement Learning","date":"2019-12-05","arxiv_id":"1912.02470","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-credit-assignment-1","slug":"hindsight-credit-assignment-1","title":"Hindsight Credit Assignment","date":"2019-12-05","arxiv_id":"1912.02503","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-objectives-by-evaluating","slug":"learning-human-objectives-by-evaluating","title":"Learning Human Objectives by Evaluating Hypothetical Behavior","date":"2019-12-05","arxiv_id":"1912.05652","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-human-objectives-by-evaluating#ran","syntology_url":"https://syntology.ai/paper/1912.05652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05652"}},"official":null}},{"url":"/paper/adaptive-online-planning-for-continual-1","slug":"adaptive-online-planning-for-continual-1","title":"Adaptive Online Planning for Continual Lifelong Learning","date":"2019-12-03","arxiv_id":"1912.01188","repositories_listed":1,"syntology":null},{"url":"/paper/mo-states-mo-problems-emergency-stop-1","slug":"mo-states-mo-problems-emergency-stop-1","title":"Mo' States Mo' Problems: Emergency Stop Mechanisms from Observation","date":"2019-12-03","arxiv_id":"1912.01649","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-farsighted-agents-tend-to-seek-power","slug":"optimal-farsighted-agents-tend-to-seek-power","title":"Optimal Policies Tend to Seek Power","date":"2019-12-03","arxiv_id":"1912.01683","repositories_listed":1,"syntology":null},{"url":"/paper/risk-averse-action-selection-using-extreme","slug":"risk-averse-action-selection-using-extreme","title":"Risk-Averse Action Selection Using Extreme Value Theory Estimates of the CVaR","date":"2019-12-03","arxiv_id":"1912.01718","repositories_listed":1,"syntology":null},{"url":"/paper/safelife-10-exploring-side-effects-in-complex","slug":"safelife-10-exploring-side-effects-in-complex","title":"SafeLife 1.0: Exploring Side Effects in Complex Environments","date":"2019-12-03","arxiv_id":"1912.01217","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safelife-10-exploring-side-effects-in-complex#ran","syntology_url":"https://syntology.ai/paper/1912.01217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01217"}},"official":{"repos":["PartnershipOnAI/safelife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/abstract-reasoning-with-distracting-features-1","slug":"abstract-reasoning-with-distracting-features-1","title":"Abstract Reasoning with Distracting Features","date":"2019-12-02","arxiv_id":"1912.00569","repositories_listed":1,"syntology":null},{"url":"/paper/a-model-based-reinforcement-learning-with-1","slug":"a-model-based-reinforcement-learning-with-1","title":"A Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-auxiliary-task-weighting-for","slug":"adaptive-auxiliary-task-weighting-for","title":"Adaptive Auxiliary Task Weighting for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/better-exploration-with-optimistic-actor-1","slug":"better-exploration-with-optimistic-actor-1","title":"Better Exploration with Optimistic Actor Critic","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-guided-hindsight-experience-replay","slug":"curriculum-guided-hindsight-experience-replay","title":"Curriculum-guided Hindsight Experience Replay","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/domes-to-drones-self-supervised-active","slug":"domes-to-drones-self-supervised-active","title":"Domes to Drones: Self-Supervised Active Triangulation for 3D Human Pose Reconstruction","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-device-placement","slug":"learning-generalizable-device-placement","title":"Learning Generalizable Device Placement Algorithms for Distributed Machine Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-local-search-heuristics-for-boolean","slug":"learning-local-search-heuristics-for-boolean","title":"Learning Local Search Heuristics for Boolean Satisfiability","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-reward-machines-for-partially","slug":"learning-reward-machines-for-partially","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/liir-learning-individual-intrinsic-reward-in","slug":"liir-learning-individual-intrinsic-reward-in","title":"LIIR: Learning Individual Intrinsic Reward in Multi-Agent Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/loaded-dice-trading-off-bias-and-variance-in-1","slug":"loaded-dice-trading-off-bias-and-variance-in-1","title":"Loaded DiCE: Trading off Bias and Variance in Any-Order Score Function Gradient Estimators for Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/non-stationary-markov-decision-processes-a-1","slug":"non-stationary-markov-decision-processes-a-1","title":"Non-Stationary Markov Decision Processes, a Worst-Case Approach using Model-Based Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/park-an-open-platform-for-learning-augmented","slug":"park-an-open-platform-for-learning-augmented","title":"Park: An Open Platform for Learning-Augmented Computer Systems","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/privacy-preserving-q-learning-with-functional","slug":"privacy-preserving-q-learning-with-functional","title":"Privacy-Preserving Q-Learning with Functional Noise in Continuous Spaces","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/propagating-uncertainty-in-reinforcement","slug":"propagating-uncertainty-in-reinforcement","title":"Propagating Uncertainty in Reinforcement Learning via Wasserstein Barycenters","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"32405e7c7d4e013b1d5bd4c005f1376ce3d5d3d0842bbc321080d497a1806f7e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}