{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/26","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":26,"pages_in_order":135,"rows_per_page":100,"rows":[2501,2600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/25","next":"/task/reinforcement-learning-2/papers/27","papers":[{"url":"/paper/open-ended-reinforcement-learning-with-neural","slug":"open-ended-reinforcement-learning-with-neural","title":"Open-Ended Reinforcement Learning with Neural Reward Functions","date":"2022-02-16","arxiv_id":"2202.08266","repositories_listed":1,"syntology":null},{"url":"/paper/soft-actor-critic-deep-reinforcement-learning","slug":"soft-actor-critic-deep-reinforcement-learning","title":"Soft Actor-Critic Deep Reinforcement Learning for Fault Tolerant Flight Control","date":"2022-02-16","arxiv_id":"2202.09262","repositories_listed":1,"syntology":null},{"url":"/paper/cup-a-conservative-update-policy-algorithm-1","slug":"cup-a-conservative-update-policy-algorithm-1","title":"CUP: A Conservative Update Policy Algorithm for Safe Reinforcement Learning","date":"2022-02-15","arxiv_id":"2202.07565","repositories_listed":1,"syntology":null},{"url":"/paper/energy-efficient-parking-analytics-system","slug":"energy-efficient-parking-analytics-system","title":"Energy-Efficient Parking Analytics System using Deep Reinforcement Learning","date":"2022-02-15","arxiv_id":"2202.08973","repositories_listed":1,"syntology":null},{"url":"/paper/graph-meta-reinforcement-learning-for","slug":"graph-meta-reinforcement-learning-for","title":"Graph Meta-Reinforcement Learning for Transferable Autonomous Mobility-on-Demand","date":"2022-02-15","arxiv_id":"2202.07147","repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-by-imagining-the-1","slug":"safe-reinforcement-learning-by-imagining-the-1","title":"Safe Reinforcement Learning by Imagining the Near Future","date":"2022-02-15","arxiv_id":"2202.07789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-by-imagining-the-1#ran","syntology_url":"https://syntology.ai/paper/2202.07789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.07789"}},"official":{"repos":["gwthomas/safe-mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-reward-models-for-cooperative","slug":"learning-reward-models-for-cooperative","title":"Learning Reward Models for Cooperative Trajectory Planning with Inverse Reinforcement Learning and Monte Carlo Tree Search","date":"2022-02-14","arxiv_id":"2202.06443","repositories_listed":1,"syntology":null},{"url":"/paper/quadsim-a-quadcopter-rotational-dynamics","slug":"quadsim-a-quadcopter-rotational-dynamics","title":"QuadSim: A Quadcopter Rotational Dynamics Simulation Framework For Reinforcement Learning Algorithms","date":"2022-02-14","arxiv_id":"2202.07021","repositories_listed":1,"syntology":null},{"url":"/paper/saute-rl-almost-surely-safe-reinforcement","slug":"saute-rl-almost-surely-safe-reinforcement","title":"Saute RL: Almost Surely Safe Reinforcement Learning Using State Augmentation","date":"2022-02-14","arxiv_id":"2202.06558","repositories_listed":1,"syntology":null},{"url":"/paper/goal-recognition-as-reinforcement-learning","slug":"goal-recognition-as-reinforcement-learning","title":"Goal Recognition as Reinforcement Learning","date":"2022-02-13","arxiv_id":"2202.06356","repositories_listed":1,"syntology":null},{"url":"/paper/learning-by-doing-controlling-a-dynamical","slug":"learning-by-doing-controlling-a-dynamical","title":"Learning by Doing: Controlling a Dynamical System using Causality, Control, and Reinforcement Learning","date":"2022-02-12","arxiv_id":"2202.06052","repositories_listed":1,"syntology":null},{"url":"/paper/choices-risks-and-reward-reports-charting","slug":"choices-risks-and-reward-reports-charting","title":"Choices, Risks, and Reward Reports: Charting Public Policy for Reinforcement Learning Systems","date":"2022-02-11","arxiv_id":"2202.05716","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-instabilities-in-variational","slug":"uncovering-instabilities-in-variational","title":"Uncovering Instabilities in Variational-Quantum Deep Q-Networks","date":"2022-02-10","arxiv_id":"2202.05195","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to-domain","slug":"a-reinforcement-learning-approach-to-domain","title":"A Reinforcement Learning Approach to Domain-Knowledge Inclusion Using Grammar Guided Symbolic Regression","date":"2022-02-09","arxiv_id":"2202.04367","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-nonparametrics-for-offline-skill","slug":"bayesian-nonparametrics-for-offline-skill","title":"Bayesian Nonparametrics for Offline Skill Discovery","date":"2022-02-09","arxiv_id":"2202.04675","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bayesian-nonparametrics-for-offline-skill#ran","syntology_url":"https://syntology.ai/paper/2202.04675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04675"}},"official":{"repos":["layer6ai-labs/bnpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/contextualize-me-the-case-for-context-in","slug":"contextualize-me-the-case-for-context-in","title":"Contextualize Me -- The Case for Context in Reinforcement Learning","date":"2022-02-09","arxiv_id":"2202.04500","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextualize-me-the-case-for-context-in#ran","syntology_url":"https://syntology.ai/paper/2202.04500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04500"}},"official":{"repos":["automl/CARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-sparse-rewards-1","slug":"reinforcement-learning-with-sparse-rewards-1","title":"Reinforcement Learning with Sparse Rewards using Guidance from Offline Demonstration","date":"2022-02-09","arxiv_id":"2202.04628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-sparse-rewards-1#ran","syntology_url":"https://syntology.ai/paper/2202.04628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04628"}},"official":{"repos":["desikrengarajan/logo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/approximating-gradients-for-differentiable","slug":"approximating-gradients-for-differentiable","title":"Approximating Gradients for Differentiable Quality Diversity in Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03666","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/approximating-gradients-for-differentiable#ran","syntology_url":"https://syntology.ai/paper/2202.03666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03666"}},"official":{"repos":["icaros-usc/dqd-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bingham-policy-parameterization-for-3d","slug":"bingham-policy-parameterization-for-3d","title":"Bingham Policy Parameterization for 3D Rotations in Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03957","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-warfarin-dosing-using-deep","slug":"optimizing-warfarin-dosing-using-deep","title":"Optimizing Warfarin Dosing using Deep Reinforcement Learning","date":"2022-02-07","arxiv_id":"2202.03486","repositories_listed":1,"syntology":null},{"url":"/paper/learning-synthetic-environments-and-reward-1","slug":"learning-synthetic-environments-and-reward-1","title":"Learning Synthetic Environments and Reward Networks for Reinforcement Learning","date":"2022-02-06","arxiv_id":"2202.02790","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-synthetic-environments-and-reward-1#ran","syntology_url":"https://syntology.ai/paper/2202.02790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02790"}},"official":{"repos":["automl/learning_environments"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-approximate-symbolic-models-for","slug":"leveraging-approximate-symbolic-models-for","title":"Leveraging Approximate Symbolic Models for Reinforcement Learning via Skill Diversity","date":"2022-02-06","arxiv_id":"2202.02886","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-multi-item","slug":"reinforcement-learning-for-multi-item","title":"Reinforcement learning for multi-item retrieval in the puzzle-based storage system","date":"2022-02-05","arxiv_id":"2202.03424","repositories_listed":1,"syntology":null},{"url":"/paper/transfer-reinforcement-learning-for-differing","slug":"transfer-reinforcement-learning-for-differing","title":"Transfer Reinforcement Learning for Differing Action Spaces via Q-Network Representations","date":"2022-02-05","arxiv_id":"2202.02442","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-sequential-experimental-design","slug":"optimizing-sequential-experimental-design","title":"Optimizing Sequential Experimental Design with Deep Reinforcement Learning","date":"2022-02-02","arxiv_id":"2202.00821","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-deep-reinforcement-learning-for","slug":"accelerating-deep-reinforcement-learning-for","title":"Accelerating Deep Reinforcement Learning for Digital Twin Network Optimization with Evolutionary Strategies","date":"2022-02-01","arxiv_id":"2202.00360","repositories_listed":1,"syntology":null},{"url":"/paper/cic-contrastive-intrinsic-control-for-1","slug":"cic-contrastive-intrinsic-control-for-1","title":"CIC: Contrastive Intrinsic Control for Unsupervised Skill Discovery","date":"2022-02-01","arxiv_id":"2202.00161","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cic-contrastive-intrinsic-control-for-1#ran","syntology_url":"https://syntology.ai/paper/2202.00161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.00161"}},"official":null}},{"url":"/paper/distributional-reinforcement-learning-via","slug":"distributional-reinforcement-learning-via","title":"Distributional Reinforcement Learning with Regularized Wasserstein Loss","date":"2022-02-01","arxiv_id":"2202.00769","repositories_listed":1,"syntology":null},{"url":"/paper/tutorial-on-amortized-optimization-for","slug":"tutorial-on-amortized-optimization-for","title":"Tutorial on amortized optimization","date":"2022-02-01","arxiv_id":"2202.00665","repositories_listed":1,"syntology":null},{"url":"/paper/dns-determinantal-point-process-based-neural","slug":"dns-determinantal-point-process-based-neural","title":"DNS: Determinantal Point Process Based Neural Network Sampler for Ensemble Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13357","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dns-determinantal-point-process-based-neural#ran","syntology_url":"https://syntology.ai/paper/2201.13357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13357"}},"official":{"repos":["IntelLabs/DNS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/don-t-change-the-algorithm-change-the-data","slug":"don-t-change-the-algorithm-change-the-data","title":"Don't Change the Algorithm, Change the Data: Exploratory Data for Offline Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13425","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/don-t-change-the-algorithm-change-the-data#ran","syntology_url":"https://syntology.ai/paper/2201.13425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13425"}},"official":{"repos":["denisyarats/exorl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reinforcement-learning-in-block","slug":"efficient-reinforcement-learning-in-block","title":"Efficient Reinforcement Learning in Block MDPs: A Model-free Representation Learning Approach","date":"2022-01-31","arxiv_id":"2202.00063","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolution-based-deep-reinforcement","slug":"graph-convolution-based-deep-reinforcement","title":"Graph Convolution-Based Deep Reinforcement Learning for Multi-Agent Decision-Making in Mixed Traffic Environments","date":"2022-01-30","arxiv_id":"2201.12776","repositories_listed":1,"syntology":null},{"url":"/paper/bellman-meets-hawkes-model-based","slug":"bellman-meets-hawkes-model-based","title":"Bellman Meets Hawkes: Model-Based Reinforcement Learning via Temporal Point Processes","date":"2022-01-29","arxiv_id":"2201.12569","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-reinforcement-learning-policies","slug":"explaining-reinforcement-learning-policies","title":"Explaining Reinforcement Learning Policies through Counterfactual Trajectories","date":"2022-01-29","arxiv_id":"2201.12462","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/explaining-reinforcement-learning-policies#ran","syntology_url":"https://syntology.ai/paper/2201.12462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12462"}},"official":{"repos":["juliusfrost/cfrl-rllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/can-wikipedia-help-offline-reinforcement","slug":"can-wikipedia-help-offline-reinforcement","title":"Can Wikipedia Help Offline Reinforcement Learning?","date":"2022-01-28","arxiv_id":"2201.12122","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-wikipedia-help-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2201.12122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12122"}},"official":{"repos":["machelreid/can-wikipedia-help-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fcmnet-full-communication-memory-net-fcmnet","slug":"fcmnet-full-communication-memory-net-fcmnet","title":"FCMNet: Full Communication Memory Net for Team-Level Cooperation in Multi-Agent Systems","date":"2022-01-28","arxiv_id":"2201.11994","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-class-abstraction-for-commonsense","slug":"leveraging-class-abstraction-for-commonsense","title":"Leveraging class abstraction for commonsense reinforcement learning via residual policy gradient methods","date":"2022-01-28","arxiv_id":"2201.12126","repositories_listed":1,"syntology":null},{"url":"/paper/mask-based-latent-reconstruction-for","slug":"mask-based-latent-reconstruction-for","title":"Mask-based Latent Reconstruction for Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.12096","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":4,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mask-based-latent-reconstruction-for#ran","syntology_url":"https://syntology.ai/paper/2201.12096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12096"}},"official":{"repos":["microsoft/Mask-based-Latent-Reconstruction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/towards-safe-reinforcement-learning-with-a","slug":"towards-safe-reinforcement-learning-with-a","title":"Towards Safe Reinforcement Learning with a Safety Editor Policy","date":"2022-01-28","arxiv_id":"2201.12427","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-safe-reinforcement-learning-with-a#ran","syntology_url":"https://syntology.ai/paper/2201.12427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12427"}},"official":{"repos":["hnyu/seditor"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/exploration-with-a-finite-brain","slug":"exploration-with-a-finite-brain","title":"Modeling Human Exploration Through Resource-Rational Reinforcement Learning","date":"2022-01-27","arxiv_id":"2201.11817","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploration-with-a-finite-brain#ran","syntology_url":"https://syntology.ai/paper/2201.11817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11817"}},"official":{"repos":["marcelbinz/resource-rational-reinforcement-learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-learning-dynamics-in-rl-using","slug":"rethinking-learning-dynamics-in-rl-using","title":"Boosting Exploration in Multi-Task Reinforcement Learning using Adversarial Networks","date":"2022-01-27","arxiv_id":"2201.11783","repositories_listed":1,"syntology":null},{"url":"/paper/moolib-a-platform-for-distributed-rl","slug":"moolib-a-platform-for-distributed-rl","title":"moolib: A Platform for Distributed RL","date":"2022-01-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/constrained-policy-optimization-via-bayesian-1","slug":"constrained-policy-optimization-via-bayesian-1","title":"Constrained Policy Optimization via Bayesian World Models","date":"2022-01-24","arxiv_id":"2201.09802","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/constrained-policy-optimization-via-bayesian-1#ran","syntology_url":"https://syntology.ai/paper/2201.09802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.09802"}},"official":{"repos":["yardenas/la-mbda"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-planning-for-temporally-1","slug":"generative-planning-for-temporally-1","title":"Generative Planning for Temporally Coordinated Exploration in Reinforcement Learning","date":"2022-01-24","arxiv_id":"2201.09765","repositories_listed":1,"syntology":null},{"url":"/paper/pearl-parallel-evolutionary-and-reinforcement","slug":"pearl-parallel-evolutionary-and-reinforcement","title":"Pearl: Parallel Evolutionary and Reinforcement Learning Library","date":"2022-01-24","arxiv_id":"2201.09568","repositories_listed":1,"syntology":null},{"url":"/paper/the-paradox-of-choice-using-attention-in","slug":"the-paradox-of-choice-using-attention-in","title":"The Paradox of Choice: Using Attention in Hierarchical Reinforcement Learning","date":"2022-01-24","arxiv_id":"2201.09653","repositories_listed":1,"syntology":null},{"url":"/paper/bag-of-tricks-for-natural-policy-gradient","slug":"bag-of-tricks-for-natural-policy-gradient","title":"Understanding the Effects of Second-Order Approximations in Natural Policy Gradient Reinforcement Learning","date":"2022-01-22","arxiv_id":"2201.09104","repositories_listed":1,"syntology":null},{"url":"/paper/environment-generation-for-zero-shot-1","slug":"environment-generation-for-zero-shot-1","title":"Environment Generation for Zero-Shot Compositional Reinforcement Learning","date":"2022-01-21","arxiv_id":"2201.08896","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/environment-generation-for-zero-shot-1#ran","syntology_url":"https://syntology.ai/paper/2201.08896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08896"}},"official":{"repos":["google-research/google-research"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tensor-and-matrix-low-rank-value-function","slug":"tensor-and-matrix-low-rank-value-function","title":"Tensor and Matrix Low-Rank Value-Function Approximation in Reinforcement Learning","date":"2022-01-21","arxiv_id":"2201.09736","repositories_listed":1,"syntology":null},{"url":"/paper/goal-conditioned-reinforcement-learning","slug":"goal-conditioned-reinforcement-learning","title":"Goal-Conditioned Reinforcement Learning: Problems and Solutions","date":"2022-01-20","arxiv_id":"2201.08299","repositories_listed":1,"syntology":null},{"url":"/paper/two-sample-testing-in-reinforcement-learning","slug":"two-sample-testing-in-reinforcement-learning","title":"Addressing Maximization Bias in Reinforcement Learning with Two-Sample Testing","date":"2022-01-20","arxiv_id":"2201.08078","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-textbook","slug":"reinforcement-learning-textbook","title":"Reinforcement Learning Textbook","date":"2022-01-19","arxiv_id":"2201.09746","repositories_listed":1,"syntology":null},{"url":"/paper/comparing-model-free-and-model-based","slug":"comparing-model-free-and-model-based","title":"Comparing Model-free and Model-based Algorithms for Offline Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05433","repositories_listed":1,"syntology":null},{"url":"/paper/smart-magnetic-microrobots-learn-to-swim-with","slug":"smart-magnetic-microrobots-learn-to-swim-with","title":"Smart Magnetic Microrobots Learn to Swim with Deep Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05599","repositories_listed":1,"syntology":null},{"url":"/paper/solving-dynamic-graph-problems-with-multi","slug":"solving-dynamic-graph-problems-with-multi","title":"Solving Dynamic Graph Problems with Multi-Attention Deep Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.04895","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-scene-text-detection-using","slug":"weakly-supervised-scene-text-detection-using","title":"Weakly Supervised Scene Text Detection using Deep Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.04866","repositories_listed":1,"syntology":null},{"url":"/paper/agent-temporal-attention-for-reward","slug":"agent-temporal-attention-for-reward","title":"Agent-Temporal Attention for Reward Redistribution in Episodic Multi-Agent Reinforcement Learning","date":"2022-01-12","arxiv_id":"2201.04612","repositories_listed":1,"syntology":null},{"url":"/paper/verified-probabilistic-policies-for-deep","slug":"verified-probabilistic-policies-for-deep","title":"Verified Probabilistic Policies for Deep Reinforcement Learning","date":"2022-01-10","arxiv_id":"2201.03698","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-5","slug":"sample-efficient-deep-reinforcement-learning-5","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","date":"2022-01-05","arxiv_id":"2201.01666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-deep-reinforcement-learning-5#ran","syntology_url":"https://syntology.ai/paper/2201.01666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01666"}},"official":{"repos":["montrealrobotics/iv_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-simulation-optimization-to-improve-zero","slug":"using-simulation-optimization-to-improve-zero","title":"Using Simulation Optimization to Improve Zero-shot Policy Transfer of Quadrotors","date":"2022-01-04","arxiv_id":"2201.01369","repositories_listed":1,"syntology":null},{"url":"/paper/settling-the-bias-and-variance-of-meta","slug":"settling-the-bias-and-variance-of-meta","title":"A Theoretical Understanding of Gradient Bias in Meta-Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15400","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/settling-the-bias-and-variance-of-meta#ran","syntology_url":"https://syntology.ai/paper/2112.15400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.15400"}},"official":{"repos":["Benjamin-eecs/Theoretical-GMRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/constraint-sampling-reinforcement-learning","slug":"constraint-sampling-reinforcement-learning","title":"Constraint Sampling Reinforcement Learning: Incorporating Expertise For Faster Learning","date":"2021-12-30","arxiv_id":"2112.15221","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-episodic-control","slug":"sequential-episodic-control","title":"Sequential memory improves sample and memory efficiency in Episodic Control","date":"2021-12-29","arxiv_id":"2112.14734","repositories_listed":1,"syntology":null},{"url":"/paper/exponential-family-model-based-reinforcement","slug":"exponential-family-model-based-reinforcement","title":"Exponential Family Model-Based Reinforcement Learning via Score Matching","date":"2021-12-28","arxiv_id":"2112.14195","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exponential-family-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.14195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.14195"}},"official":{"repos":["anmolkabra/score-matching-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-performance-of-backward-chained","slug":"improving-the-performance-of-backward-chained","title":"Improving the Performance of Backward Chained Behavior Trees that use Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13744","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-dynamic-convex","slug":"reinforcement-learning-with-dynamic-convex","title":"Reinforcement Learning with Dynamic Convex Risk Measures","date":"2021-12-26","arxiv_id":"2112.13414","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-dynamic-convex#ran","syntology_url":"https://syntology.ai/paper/2112.13414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.13414"}},"official":{"repos":["acoache/rl-dynamicconvexrisk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-walk-with-dual-agents-for","slug":"learning-to-walk-with-dual-agents-for","title":"Learning to Walk with Dual Agents for Knowledge Graph Reasoning","date":"2021-12-23","arxiv_id":"2112.12876","repositories_listed":1,"syntology":null},{"url":"/paper/safety-and-liveness-guarantees-through-reach","slug":"safety-and-liveness-guarantees-through-reach","title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12288","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-for-11","slug":"a-deep-reinforcement-learning-approach-for-11","title":"A Deep Reinforcement Learning Approach for Solving the Traveling Salesman Problem with Drone","date":"2021-12-22","arxiv_id":"2112.12545","repositories_listed":1,"syntology":null},{"url":"/paper/alpha-mini-minichess-agent-with-deep","slug":"alpha-mini-minichess-agent-with-deep","title":"Alpha-Mini: Minichess Agent with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.13666","repositories_listed":1,"syntology":null},{"url":"/paper/direct-behavior-specification-via-constrained","slug":"direct-behavior-specification-via-constrained","title":"Direct Behavior Specification via Constrained Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.12228","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-robustness-of-deep","slug":"evaluating-the-robustness-of-deep","title":"Evaluating the Robustness of Deep Reinforcement Learning for Autonomous Policies in a Multi-agent Urban Driving Environment","date":"2021-12-22","arxiv_id":"2112.11947","repositories_listed":1,"syntology":null},{"url":"/paper/newsvendor-model-with-deep-reinforcement","slug":"newsvendor-model-with-deep-reinforcement","title":"Newsvendor Model with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.12544","repositories_listed":1,"syntology":null},{"url":"/paper/variational-quantum-soft-actor-critic","slug":"variational-quantum-soft-actor-critic","title":"Variational Quantum Soft Actor-Critic","date":"2021-12-20","arxiv_id":"2112.11921","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-expert-guided-symmetry-detection","slug":"exploiting-expert-guided-symmetry-detection","title":"Data Augmentation through Expert-guided Symmetry Detection to Improve Performance in Offline Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09943","repositories_listed":1,"syntology":null},{"url":"/paper/space-non-cooperative-object-active-tracking","slug":"space-non-cooperative-object-active-tracking","title":"Space Non-cooperative Object Active Tracking with Deep Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09854","repositories_listed":1,"syntology":null},{"url":"/paper/inherently-explainable-reinforcement-learning","slug":"inherently-explainable-reinforcement-learning","title":"Inherently Explainable Reinforcement Learning in Natural Language","date":"2021-12-16","arxiv_id":"2112.08907","repositories_listed":1,"syntology":null},{"url":"/paper/feature-attending-recurrent-modules-for","slug":"feature-attending-recurrent-modules-for","title":"Feature-Attending Recurrent Modules for Generalization in Reinforcement Learning","date":"2021-12-15","arxiv_id":"2112.08369","repositories_listed":1,"syntology":null},{"url":"/paper/conjugated-discrete-distributions-for","slug":"conjugated-discrete-distributions-for","title":"Conjugated Discrete Distributions for Distributional Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07424","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-and-adaptive-penalty-for-model","slug":"conservative-and-adaptive-penalty-for-model","title":"Conservative and Adaptive Penalty for Model-Based Safe Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07701","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-actor-executor-critic-for-image-to","slug":"stochastic-actor-executor-critic-for-image-to","title":"Stochastic Actor-Executor-Critic for Image-to-Image Translation","date":"2021-12-14","arxiv_id":"2112.07403","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-planner-actor-critic-for","slug":"stochastic-planner-actor-critic-for","title":"Stochastic Planner-Actor-Critic for Unsupervised Deformable Image Registration","date":"2021-12-14","arxiv_id":"2112.07415","repositories_listed":1,"syntology":null},{"url":"/paper/finrl-meta-a-universe-of-near-real-market","slug":"finrl-meta-a-universe-of-near-real-market","title":"FinRL-Meta: A Universe of Near-Real Market Environments for Data-Driven Deep Reinforcement Learning in Quantitative Finance","date":"2021-12-13","arxiv_id":"2112.06753","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/finrl-meta-a-universe-of-near-real-market#ran","syntology_url":"https://syntology.ai/paper/2112.06753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06753"}},"official":{"repos":["ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/human-level-control-through-directly-trained","slug":"human-level-control-through-directly-trained","title":"Human-Level Control through Directly-Trained Deep Spiking Q-Networks","date":"2021-12-13","arxiv_id":"2201.07211","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/human-level-control-through-directly-trained#ran","syntology_url":"https://syntology.ai/paper/2201.07211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.07211"}},"official":{"repos":["aptx395/deep-spiking-q-networks"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pantheonrl-a-marl-library-for-dynamic","slug":"pantheonrl-a-marl-library-for-dynamic","title":"PantheonRL: A MARL Library for Dynamic Training Interactions","date":"2021-12-13","arxiv_id":"2112.07013","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pantheonrl-a-marl-library-for-dynamic#ran","syntology_url":"https://syntology.ai/paper/2112.07013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.07013"}},"official":{"repos":["Stanford-ILIAD/PantheonRL"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tree-based-focused-web-crawling-with","slug":"tree-based-focused-web-crawling-with","title":"Tree-based Focused Web Crawling with Reinforcement Learning","date":"2021-12-12","arxiv_id":"2112.07620","repositories_listed":1,"syntology":null},{"url":"/paper/elegantrl-podracer-scalable-and-elastic","slug":"elegantrl-podracer-scalable-and-elastic","title":"ElegantRL-Podracer: Scalable and Elastic Library for Cloud-Native Deep Reinforcement Learning","date":"2021-12-11","arxiv_id":"2112.05923","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/elegantrl-podracer-scalable-and-elastic#ran","syntology_url":"https://syntology.ai/paper/2112.05923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05923"}},"official":{"repos":["ai4finance-foundation/elegantrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ostrichrl-a-musculoskeletal-ostrich","slug":"ostrichrl-a-musculoskeletal-ostrich","title":"OstrichRL: A Musculoskeletal Ostrich Simulation to Study Bio-mechanical Locomotion","date":"2021-12-11","arxiv_id":"2112.06061","repositories_listed":1,"syntology":null},{"url":"/paper/blockwise-sequential-model-learning-for","slug":"blockwise-sequential-model-learning-for","title":"Blockwise Sequential Model Learning for Partially Observable Reinforcement Learning","date":"2021-12-10","arxiv_id":"2112.05343","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-network-with-proximal-iteration-1","slug":"deep-q-network-with-proximal-iteration-1","title":"Faster Deep Reinforcement Learning with Slower Online Network","date":"2021-12-10","arxiv_id":"2112.05848","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-q-network-with-proximal-iteration-1#ran","syntology_url":"https://syntology.ai/paper/2112.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05848"}},"official":{"repos":["amazon-research/fast-rl-with-slow-updates"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-multiple-gaits-of-quadruped-robot","slug":"learning-multiple-gaits-of-quadruped-robot","title":"Learning multiple gaits of quadruped robot using hierarchical reinforcement learning","date":"2021-12-09","arxiv_id":"2112.04741","repositories_listed":1,"syntology":null},{"url":"/paper/value-function-factorisation-with-hypergraph","slug":"value-function-factorisation-with-hypergraph","title":"Cooperative Multi-Agent Reinforcement Learning with Hypergraph Convolution","date":"2021-12-09","arxiv_id":"2112.06771","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-model-and-deep-reinforcement","slug":"attention-based-model-and-deep-reinforcement","title":"Attention-Based Model and Deep Reinforcement Learning for Distribution of Event Processing Tasks","date":"2021-12-07","arxiv_id":"2112.03835","repositories_listed":1,"syntology":null},{"url":"/paper/federated-deep-reinforcement-learning-for-the","slug":"federated-deep-reinforcement-learning-for-the","title":"Federated Deep Reinforcement Learning for the Distributed Control of NextG Wireless Networks","date":"2021-12-07","arxiv_id":"2112.03465","repositories_listed":1,"syntology":null},{"url":"/paper/godot-reinforcement-learning-agents","slug":"godot-reinforcement-learning-agents","title":"Godot Reinforcement Learning Agents","date":"2021-12-07","arxiv_id":"2112.03636","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/godot-reinforcement-learning-agents#ran","syntology_url":"https://syntology.ai/paper/2112.03636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03636"}},"official":{"repos":["edbeeching/godot_rl_agents"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/flexible-option-learning-1","slug":"flexible-option-learning-1","title":"Flexible Option Learning","date":"2021-12-06","arxiv_id":"2112.03097","repositories_listed":1,"syntology":null},{"url":"/paper/functional-regularization-for-reinforcement-1","slug":"functional-regularization-for-reinforcement-1","title":"Functional Regularization for Reinforcement Learning via Learned Fourier Features","date":"2021-12-06","arxiv_id":"2112.03257","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/functional-regularization-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2112.03257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03257"}},"official":{"repos":["alexlioralexli/learned-fourier-features"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-reinforcement-learning-with-5","slug":"hierarchical-reinforcement-learning-with-5","title":"Hierarchical Reinforcement Learning with Timed Subgoals","date":"2021-12-06","arxiv_id":"2112.03100","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/hierarchical-reinforcement-learning-with-5#ran","syntology_url":"https://syntology.ai/paper/2112.03100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03100"}},"official":{"repos":["martius-lab/hits"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}}],"record_sha256":"63ae77031f4f0a5401a43c7274497906311ea087eff853f9015fa076dc87524f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}