{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/24","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":24,"pages_in_order":135,"rows_per_page":100,"rows":[2301,2400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/23","next":"/task/reinforcement-learning-2/papers/25","papers":[{"url":"/paper/interactive-query-assisted-summarization-via","slug":"interactive-query-assisted-summarization-via","title":"Interactive Query-Assisted Summarization via Deep Reinforcement Learning","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-natural-language-generation-with","slug":"learning-natural-language-generation-with","title":"Learning Natural Language Generation with Truncated Reinforcement Learning","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lifelong-inverse-reinforcement-learning-1","slug":"lifelong-inverse-reinforcement-learning-1","title":"Lifelong Inverse Reinforcement Learning","date":"2022-07-01","arxiv_id":"2207.00461","repositories_listed":1,"syntology":null},{"url":"/paper/modular-lifelong-reinforcement-learning-via-1","slug":"modular-lifelong-reinforcement-learning-via-1","title":"Modular Lifelong Reinforcement Learning via Neural Composition","date":"2022-07-01","arxiv_id":"2207.00429","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-lifelong-reinforcement-learning-via-1#ran","syntology_url":"https://syntology.ai/paper/2207.00429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00429"}},"official":{"repos":["lifelong-ml/mendez2022modularlifelongrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","repositories_listed":1,"syntology":null},{"url":"/paper/mastering-the-game-of-stratego-with-model","slug":"mastering-the-game-of-stratego-with-model","title":"Mastering the Game of Stratego with Model-Free Multiagent Reinforcement Learning","date":"2022-06-30","arxiv_id":"2206.15378","repositories_listed":1,"syntology":null},{"url":"/paper/conditionally-elicitable-dynamic-risk","slug":"conditionally-elicitable-dynamic-risk","title":"Conditionally Elicitable Dynamic Risk Measures for Deep Reinforcement Learning","date":"2022-06-29","arxiv_id":"2206.14666","repositories_listed":1,"syntology":null},{"url":"/paper/daydreamer-world-models-for-physical-robot","slug":"daydreamer-world-models-for-physical-robot","title":"DayDreamer: World Models for Physical Robot Learning","date":"2022-06-28","arxiv_id":"2206.14176","repositories_listed":1,"syntology":null},{"url":"/paper/distspectrl-distributing-specifications-in","slug":"distspectrl-distributing-specifications-in","title":"DistSPECTRL: Distributing Specifications in Multi-Agent Reinforcement Learning Systems","date":"2022-06-28","arxiv_id":"2206.13754","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-building-control","slug":"zero-shot-building-control","title":"Low Emission Building Control with Zero-Shot Reinforcement Learning","date":"2022-06-28","arxiv_id":"2206.14191","repositories_listed":1,"syntology":null},{"url":"/paper/distinguishing-learning-rules-with-brain","slug":"distinguishing-learning-rules-with-brain","title":"Distinguishing Learning Rules with Brain Machine Interfaces","date":"2022-06-27","arxiv_id":"2206.13448","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distinguishing-learning-rules-with-brain#ran","syntology_url":"https://syntology.ai/paper/2206.13448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13448"}},"official":{"repos":["jacobfulano/learning-rules-with-bmi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-trust-your-simulator-dynamics-aware","slug":"when-to-trust-your-simulator-dynamics-aware","title":"When to Trust Your Simulator: Dynamics-Aware Hybrid Offline-and-Online Reinforcement Learning","date":"2022-06-27","arxiv_id":"2206.13464","repositories_listed":1,"syntology":null},{"url":"/paper/guided-exploration-in-reinforcement-learning","slug":"guided-exploration-in-reinforcement-learning","title":"Guided Exploration in Reinforcement Learning via Monte Carlo Critic Optimization","date":"2022-06-25","arxiv_id":"2206.12674","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-adaptive-1","slug":"reinforcement-learning-based-adaptive-1","title":"Reinforcement learning based adaptive metaheuristics","date":"2022-06-24","arxiv_id":"2206.12233","repositories_listed":1,"syntology":null},{"url":"/paper/learning-representations-for-control-with","slug":"learning-representations-for-control-with","title":"Multi-Horizon Representations with Hierarchical Forward Models for Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11396","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-car-parking-using-reinforcement","slug":"multi-agent-car-parking-using-reinforcement","title":"Multi-Agent Car Parking using Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.13338","repositories_listed":1,"syntology":null},{"url":"/paper/pac-assisted-value-factorisation-with","slug":"pac-assisted-value-factorisation-with","title":"PAC: Assisted Value Factorisation with Counterfactual Predictions in Multi-Agent Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11420","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pac-assisted-value-factorisation-with#ran","syntology_url":"https://syntology.ai/paper/2206.11420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11420"}},"official":{"repos":["hanhananderson/pac-marl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-reinforcement-learning-with-finite","slug":"meta-reinforcement-learning-with-finite","title":"Meta Reinforcement Learning with Finite Training Tasks -- a Density Estimation Approach","date":"2022-06-21","arxiv_id":"2206.10716","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-through-1","slug":"robust-deep-reinforcement-learning-through-1","title":"Robust Deep Reinforcement Learning through Bootstrapped Opportunistic Curriculum","date":"2022-06-21","arxiv_id":"2206.10057","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-1#ran","syntology_url":"https://syntology.ai/paper/2206.10057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10057"}},"official":{"repos":["jlwu002/bcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-task-representations-for-offline-meta","slug":"robust-task-representations-for-offline-meta","title":"Robust Task Representations for Offline Meta-Reinforcement Learning via Contrastive Learning","date":"2022-06-21","arxiv_id":"2206.10442","repositories_listed":1,"syntology":null},{"url":"/paper/maser-multi-agent-reinforcement-learning-with","slug":"maser-multi-agent-reinforcement-learning-with","title":"MASER: Multi-Agent Reinforcement Learning with Subgoals Generated from Experience Replay Buffer","date":"2022-06-20","arxiv_id":"2206.10607","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-efficient-deep-reinforcement","slug":"sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","repositories_listed":1,"syntology":null},{"url":"/paper/deep-inverse-reinforcement-learning-for-route","slug":"deep-inverse-reinforcement-learning-for-route","title":"A deep inverse reinforcement learning approach to route choice modeling with context-dependent rewards","date":"2022-06-18","arxiv_id":"2206.10598","repositories_listed":1,"syntology":null},{"url":"/paper/fast-population-based-reinforcement-learning","slug":"fast-population-based-reinforcement-learning","title":"Fast Population-Based Reinforcement Learning on a Single Machine","date":"2022-06-17","arxiv_id":"2206.08888","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-population-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.08888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08888"}},"official":null}},{"url":"/paper/logic-based-reward-shaping-for-multi-agent","slug":"logic-based-reward-shaping-for-multi-agent","title":"Logic-based Reward Shaping for Multi-Agent Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08881","repositories_listed":1,"syntology":null},{"url":"/paper/saferl-kit-evaluating-efficient-reinforcement","slug":"saferl-kit-evaluating-efficient-reinforcement","title":"SafeRL-Kit: Evaluating Efficient Reinforcement Learning Methods for Safe Autonomous Driving","date":"2022-06-17","arxiv_id":"2206.08528","repositories_listed":1,"syntology":null},{"url":"/paper/smpl-simulated-industrial-manufacturing-and","slug":"smpl-simulated-industrial-manufacturing-and","title":"SMPL: Simulated Industrial Manufacturing and Process Control Learning Environments","date":"2022-06-17","arxiv_id":"2206.08851","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smpl-simulated-industrial-manufacturing-and#ran","syntology_url":"https://syntology.ai/paper/2206.08851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08851"}},"official":{"repos":["smpl-env/smpl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-state-of-sparse-training-in-deep","slug":"the-state-of-sparse-training-in-deep","title":"The State of Sparse Training in Deep Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.10369","repositories_listed":1,"syntology":null},{"url":"/paper/towards-human-level-bimanual-dexterous","slug":"towards-human-level-bimanual-dexterous","title":"Towards Human-Level Bimanual Dexterous Manipulation with Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08686","repositories_listed":1,"syntology":null},{"url":"/paper/barrier-certified-safety-learning-control","slug":"barrier-certified-safety-learning-control","title":"Barrier Certified Safety Learning Control: When Sum-of-Square Programming Meets Reinforcement Learning","date":"2022-06-16","arxiv_id":"2206.07915","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-shared","slug":"reinforcement-learning-enhanced-shared","title":"Reinforcement Learning-enhanced Shared-account Cross-domain Sequential Recommendation","date":"2022-06-16","arxiv_id":"2206.08088","repositories_listed":1,"syntology":null},{"url":"/paper/search-based-testing-approach-for-deep","slug":"search-based-testing-approach-for-deep","title":"A Search-Based Testing Approach for Deep Reinforcement Learning Agents","date":"2022-06-15","arxiv_id":"2206.07813","repositories_listed":1,"syntology":null},{"url":"/paper/training-discrete-deep-generative-models-via","slug":"training-discrete-deep-generative-models-via","title":"Training Discrete Deep Generative Models via Gapped Straight-Through Estimator","date":"2022-06-15","arxiv_id":"2206.07235","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-discrete-deep-generative-models-via#ran","syntology_url":"https://syntology.ai/paper/2206.07235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07235"}},"official":{"repos":["chijames/gst"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/defending-observation-attacks-in-deep","slug":"defending-observation-attacks-in-deep","title":"Defending Observation Attacks in Deep Reinforcement Learning via Detection and Denoising","date":"2022-06-14","arxiv_id":"2206.07188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/defending-observation-attacks-in-deep#ran","syntology_url":"https://syntology.ai/paper/2206.07188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07188"}},"official":{"repos":["ZikangXiong/rl-detect-and-denoise-defense"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-meta-reinforcement-learners-1","slug":"transformers-are-meta-reinforcement-learners-1","title":"Transformers are Meta-Reinforcement Learners","date":"2022-06-14","arxiv_id":"2206.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transformers-are-meta-reinforcement-learners-1#ran","syntology_url":"https://syntology.ai/paper/2206.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06614"}},"official":{"repos":["luckeciano/transformers-metarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/universally-expressive-communication-in-multi","slug":"universally-expressive-communication-in-multi","title":"Universally Expressive Communication in Multi-Agent Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.06758","repositories_listed":1,"syntology":null},{"url":"/paper/ign-implicit-generative-networks","slug":"ign-implicit-generative-networks","title":"IGN : Implicit Generative Networks","date":"2022-06-13","arxiv_id":"2206.05860","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-placement-of","slug":"reinforcement-learning-based-placement-of","title":"Reinforcement Learning-based Placement of Charging Stations in Urban Road Networks","date":"2022-06-13","arxiv_id":"2206.06011","repositories_listed":1,"syntology":null},{"url":"/paper/case-based-inverse-reinforcement-learning","slug":"case-based-inverse-reinforcement-learning","title":"Case-Based Inverse Reinforcement Learning Using Temporal Coherence","date":"2022-06-12","arxiv_id":"2206.05827","repositories_listed":1,"syntology":null},{"url":"/paper/roi-constrained-bidding-via-curriculum-guided","slug":"roi-constrained-bidding-via-curriculum-guided","title":"ROI-Constrained Bidding via Curriculum-Guided Bayesian Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05240","repositories_listed":1,"syntology":null},{"url":"/paper/a-relational-intervention-approach-for-1","slug":"a-relational-intervention-approach-for-1","title":"A Relational Intervention Approach for Unsupervised Dynamics Generalization in Model-Based Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04551","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-reinforcement-learning-via-2","slug":"towards-safe-reinforcement-learning-via-2","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value-at-Risk","date":"2022-06-09","arxiv_id":"2206.04436","repositories_listed":1,"syntology":null},{"url":"/paper/value-memory-graph-a-graph-structured-world","slug":"value-memory-graph-a-graph-structured-world","title":"Value Memory Graph: A Graph-Structured World Model for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04384","repositories_listed":1,"syntology":null},{"url":"/paper/designing-reinforcement-learning-algorithms","slug":"designing-reinforcement-learning-algorithms","title":"Designing Reinforcement Learning Algorithms for Digital Interventions: Pre-implementation Guidelines","date":"2022-06-08","arxiv_id":"2206.03944","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-voltage-in-power-distribution","slug":"stabilizing-voltage-in-power-distribution","title":"Stabilizing Voltage in Power Distribution Networks via Multi-Agent Reinforcement Learning with Transformer","date":"2022-06-08","arxiv_id":"2206.03721","repositories_listed":1,"syntology":null},{"url":"/paper/deeptpi-test-point-insertion-with-deep","slug":"deeptpi-test-point-insertion-with-deep","title":"DeepTPI: Test Point Insertion with Deep Reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.06975","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-i-ll-go-offline-goal-conditioned","slug":"how-far-i-ll-go-offline-goal-conditioned","title":"How Far I'll Go: Offline Goal-Conditioned Reinforcement Learning via $f$-Advantage Regression","date":"2022-06-07","arxiv_id":"2206.03023","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-far-i-ll-go-offline-goal-conditioned#ran","syntology_url":"https://syntology.ai/paper/2206.03023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03023"}},"official":{"repos":["jasonma2016/gofar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/look-back-when-surprised-stabilizing-reverse","slug":"look-back-when-surprised-stabilizing-reverse","title":"Introspective Experience Replay: Look Back When Surprised","date":"2022-06-07","arxiv_id":"2206.03171","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-nav-a-library-for-neurally-plausible","slug":"neuro-nav-a-library-for-neurally-plausible","title":"Neuro-Nav: A Library for Neurally-Plausible Reinforcement Learning","date":"2022-06-06","arxiv_id":"2206.03312","repositories_listed":1,"syntology":null},{"url":"/paper/rorl-robust-offline-reinforcement-learning","slug":"rorl-robust-offline-reinforcement-learning","title":"RORL: Robust Offline Reinforcement Learning via Conservative Smoothing","date":"2022-06-06","arxiv_id":"2206.02829","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rorl-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.02829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02829"}},"official":{"repos":["yangrui2015/rorl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-tabula-rasa-reincarnating","slug":"beyond-tabula-rasa-reincarnating","title":"Reincarnating Reinforcement Learning: Reusing Prior Computation to Accelerate Progress","date":"2022-06-03","arxiv_id":"2206.01626","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-tabula-rasa-reincarnating#ran","syntology_url":"https://syntology.ai/paper/2206.01626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01626"}},"official":{"repos":["google-research/reincarnating_rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-transformer-q-networks-for-partially","slug":"deep-transformer-q-networks-for-partially","title":"Deep Transformer Q-Networks for Partially Observable Reinforcement Learning","date":"2022-06-02","arxiv_id":"2206.01078","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-transformer-q-networks-for-partially#ran","syntology_url":"https://syntology.ai/paper/2206.01078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01078"}},"official":{"repos":["kevslinger/dtqn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-does-return-conditioned-supervised","slug":"when-does-return-conditioned-supervised","title":"When does return-conditioned supervised learning work for offline reinforcement learning?","date":"2022-06-02","arxiv_id":"2206.01079","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/when-does-return-conditioned-supervised#ran","syntology_url":"https://syntology.ai/paper/2206.01079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01079"}},"official":{"repos":["davidbrandfonbrener/rcsl-paper"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dm-2-distributed-multi-agent-reinforcement","slug":"dm-2-distributed-multi-agent-reinforcement","title":"DM$^2$: Decentralized Multi-Agent Reinforcement Learning for Distribution Matching","date":"2022-06-01","arxiv_id":"2206.00233","repositories_listed":1,"syntology":null},{"url":"/paper/a-meta-reinforcement-learning-approach-for","slug":"a-meta-reinforcement-learning-approach-for","title":"A Meta Reinforcement Learning Approach for Predictive Autoscaling in the Cloud","date":"2022-05-31","arxiv_id":"2205.15795","repositories_listed":1,"syntology":null},{"url":"/paper/iglu-gridworld-simple-and-fast-environment","slug":"iglu-gridworld-simple-and-fast-environment","title":"IGLU Gridworld: Simple and Fast Environment for Embodied Dialog Agents","date":"2022-05-31","arxiv_id":"2206.00142","repositories_listed":1,"syntology":null},{"url":"/paper/dep-rl-embodied-exploration-for-reinforcement","slug":"dep-rl-embodied-exploration-for-reinforcement","title":"DEP-RL: Embodied Exploration for Reinforcement Learning in Overactuated and Musculoskeletal Systems","date":"2022-05-30","arxiv_id":"2206.00484","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dep-rl-embodied-exploration-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.00484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00484"}},"official":{"repos":["martius-lab/depRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-reward-poisoning-attacks-on-online#ran","syntology_url":"https://syntology.ai/paper/2205.14842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14842"}},"official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-is-a","slug":"multi-agent-reinforcement-learning-is-a","title":"Multi-Agent Reinforcement Learning is a Sequence Modeling Problem","date":"2022-05-30","arxiv_id":"2205.14953","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-terminator","slug":"reinforcement-learning-with-a-terminator","title":"Reinforcement Learning with a Terminator","date":"2022-05-30","arxiv_id":"2205.15376","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-a-terminator#ran","syntology_url":"https://syntology.ai/paper/2205.15376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15376"}},"official":{"repos":["guytenn/terminator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlx2-training-a-sparse-deep-reinforcement","slug":"rlx2-training-a-sparse-deep-reinforcement","title":"RLx2: Training a Sparse Deep Reinforcement Learning Model from Scratch","date":"2022-05-30","arxiv_id":"2205.15043","repositories_listed":1,"syntology":null},{"url":"/paper/learning-security-strategies-through-game","slug":"learning-security-strategies-through-game","title":"Learning Security Strategies through Game Play and Optimal Stopping","date":"2022-05-29","arxiv_id":"2205.14694","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-robustness-of-safe-reinforcement","slug":"on-the-robustness-of-safe-reinforcement","title":"On the Robustness of Safe Reinforcement Learning under Observational Perturbations","date":"2022-05-29","arxiv_id":"2205.14691","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-the-robustness-of-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14691"}},"official":{"repos":["liuzuxin/safe-rl-robustness"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-benefits-of-representational","slug":"provable-benefits-of-representational","title":"Provable Benefits of Representational Transfer in Reinforcement Learning","date":"2022-05-29","arxiv_id":"2205.14571","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-branch-and-bound","slug":"reinforcement-learning-for-branch-and-bound","title":"Reinforcement Learning for Branch-and-Bound Optimisation using Retrospective Trajectories","date":"2022-05-28","arxiv_id":"2205.14345","repositories_listed":1,"syntology":null},{"url":"/paper/fedformer-contextual-federation-with","slug":"fedformer-contextual-federation-with","title":"FedFormer: Contextual Federation with Attention in Reinforcement Learning","date":"2022-05-27","arxiv_id":"2205.13697","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedformer-contextual-federation-with#ran","syntology_url":"https://syntology.ai/paper/2205.13697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13697"}},"official":{"repos":["liamhebert/FedFormer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/drlcomplex-reconstruction-of-protein","slug":"drlcomplex-reconstruction-of-protein","title":"DRLComplex: Reconstruction of protein quaternary structures using deep reinforcement learning","date":"2022-05-26","arxiv_id":"2205.13594","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-network-reconfiguration-for-entropy","slug":"dynamic-network-reconfiguration-for-entropy","title":"Dynamic Network Reconfiguration for Entropy Maximization using Deep Reinforcement Learning","date":"2022-05-26","arxiv_id":"2205.13578","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-approach-for-mapping","slug":"reinforcement-learning-approach-for-mapping","title":"Reinforcement Learning Approach for Mapping Applications to Dataflow-Based Coarse-Grained Reconfigurable Array","date":"2022-05-26","arxiv_id":"2205.13675","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-reinforcement-adaptation-for-1","slug":"unsupervised-reinforcement-adaptation-for-1","title":"Unsupervised Reinforcement Adaptation for Class-Imbalanced Text Classification","date":"2022-05-26","arxiv_id":"2205.13139","repositories_listed":1,"syntology":null},{"url":"/paper/impartial-games-a-challenge-for-reinforcement","slug":"impartial-games-a-challenge-for-reinforcement","title":"Impartial Games: A Challenge for Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12787","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-query-internet-text-for-informing","slug":"learning-to-query-internet-text-for-informing","title":"Learning to Query Internet Text for Informing Reinforcement Learning Agents","date":"2022-05-25","arxiv_id":"2205.13079","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-knowledge-alignment-with","slug":"multimodal-knowledge-alignment-with","title":"Multimodal Knowledge Alignment with Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12630","repositories_listed":1,"syntology":null},{"url":"/paper/rlprompt-optimizing-discrete-text-prompts","slug":"rlprompt-optimizing-discrete-text-prompts","title":"RLPrompt: Optimizing Discrete Text Prompts with Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12548","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/rlprompt-optimizing-discrete-text-prompts#ran","syntology_url":"https://syntology.ai/paper/2205.12548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12548"}},"official":{"repos":["mingkaid/rl-prompt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/scalable-multi-agent-model-based","slug":"scalable-multi-agent-model-based","title":"Scalable Multi-Agent Model-Based Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.15023","repositories_listed":1,"syntology":null},{"url":"/paper/skill-machines-temporal-logic-composition-in","slug":"skill-machines-temporal-logic-composition-in","title":"Skill Machines: Temporal Logic Skill Composition in Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12532","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/skill-machines-temporal-logic-composition-in#ran","syntology_url":"https://syntology.ai/paper/2205.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12532"}},"official":{"repos":["geraudnt/skill_machines"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/tiered-reinforcement-learning-pessimism-in","slug":"tiered-reinforcement-learning-pessimism-in","title":"Tiered Reinforcement Learning: Pessimism in the Face of Uncertainty and Constant Regret","date":"2022-05-25","arxiv_id":"2205.12418","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/tiered-reinforcement-learning-pessimism-in#ran","syntology_url":"https://syntology.ai/paper/2205.12418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12418"}},"official":{"repos":["jiaweihhuang/tiered-rl-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/concurrent-credit-assignment-for-data","slug":"concurrent-credit-assignment-for-data","title":"Concurrent Credit Assignment for Data-efficient Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12020","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-class","slug":"deep-reinforcement-learning-for-multi-class","title":"Deep Reinforcement Learning for Multi-class Imbalanced Training","date":"2022-05-24","arxiv_id":"2205.12070","repositories_listed":1,"syntology":null},{"url":"/paper/meta-policy-learning-for-cold-start","slug":"meta-policy-learning-for-cold-start","title":"Meta Policy Learning for Cold-Start Conversational Recommendation","date":"2022-05-24","arxiv_id":"2205.11788","repositories_listed":1,"syntology":null},{"url":"/paper/an-evaluation-study-of-intrinsic-motivation","slug":"an-evaluation-study-of-intrinsic-motivation","title":"An Evaluation Study of Intrinsic Motivation Techniques applied to Reinforcement Learning over Hard Exploration Environments","date":"2022-05-23","arxiv_id":"2205.11184","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-evaluation-study-of-intrinsic-motivation#ran","syntology_url":"https://syntology.ai/paper/2205.11184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11184"}},"official":{"repos":["aklein1995/intrinsic_motivation_techniques_study"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/memory-efficient-reinforcement-learning-with","slug":"memory-efficient-reinforcement-learning-with","title":"Memory-efficient Reinforcement Learning with Value-based Knowledge Consolidation","date":"2022-05-22","arxiv_id":"2205.10868","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memory-efficient-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2205.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10868"}},"official":{"repos":["qlan3/MeDQN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/user-interactive-offline-reinforcement","slug":"user-interactive-offline-reinforcement","title":"User-Interactive Offline Reinforcement Learning","date":"2022-05-21","arxiv_id":"2205.10629","repositories_listed":1,"syntology":null},{"url":"/paper/a-review-of-safe-reinforcement-learning","slug":"a-review-of-safe-reinforcement-learning","title":"A Review of Safe Reinforcement Learning: Methods, Theory and Applications","date":"2022-05-20","arxiv_id":"2205.10330","repositories_listed":1,"syntology":null},{"url":"/paper/arlo-a-framework-for-automated-reinforcement","slug":"arlo-a-framework-for-automated-reinforcement","title":"ARLO: A Framework for Automated Reinforcement Learning","date":"2022-05-20","arxiv_id":"2205.10416","repositories_listed":1,"syntology":null},{"url":"/paper/self-paced-multi-agent-reinforcement-learning","slug":"self-paced-multi-agent-reinforcement-learning","title":"Learning Progress Driven Multi-Agent Curriculum","date":"2022-05-20","arxiv_id":"2205.10016","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-greedy-search-tracking-by-multi-agent","slug":"beyond-greedy-search-tracking-by-multi-agent","title":"Beyond Greedy Search: Tracking by Multi-Agent Reinforcement Learning-based Beam Search","date":"2022-05-19","arxiv_id":"2205.09676","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-time-1","slug":"deep-reinforcement-learning-for-time-1","title":"Deep Reinforcement Learning for Time Allocation and Directional Transmission in Joint Radar-Communication","date":"2022-05-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-brain-inspired","slug":"reinforcement-learning-with-brain-inspired","title":"Reinforcement Learning with Brain-Inspired Modulation can Improve Adaptation to Environmental Changes","date":"2022-05-19","arxiv_id":"2205.09729","repositories_listed":1,"syntology":null},{"url":"/paper/time-series-anomaly-detection-via","slug":"time-series-anomaly-detection-via","title":"Time Series Anomaly Detection via Reinforcement Learning-Based Model Selection","date":"2022-05-19","arxiv_id":"2205.09884","repositories_listed":1,"syntology":null},{"url":"/paper/a2c-is-a-special-case-of-ppo","slug":"a2c-is-a-special-case-of-ppo","title":"A2C is a special case of PPO","date":"2022-05-18","arxiv_id":"2205.09123","repositories_listed":1,"syntology":null},{"url":"/paper/neighborhood-mixup-experience-replay-local","slug":"neighborhood-mixup-experience-replay-local","title":"Neighborhood Mixup Experience Replay: Local Convex Interpolation for Improved Sample Efficiency in Continuous Control Tasks","date":"2022-05-18","arxiv_id":"2205.09117","repositories_listed":1,"syntology":null},{"url":"/paper/deepsim-a-reinforcement-learning-environment","slug":"deepsim-a-reinforcement-learning-environment","title":"DeepSim: A Reinforcement Learning Environment Build Toolkit for ROS and Gazebo","date":"2022-05-17","arxiv_id":"2205.08034","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-unsupervised-sentence-compression-1","slug":"efficient-unsupervised-sentence-compression-1","title":"Efficient Unsupervised Sentence Compression by Fine-tuning Transformers with Reinforcement Learning","date":"2022-05-17","arxiv_id":"2205.08221","repositories_listed":1,"syntology":null},{"url":"/paper/the-primacy-bias-in-deep-reinforcement","slug":"the-primacy-bias-in-deep-reinforcement","title":"The Primacy Bias in Deep Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07802","repositories_listed":1,"syntology":null},{"url":"/paper/unified-distributed-environment","slug":"unified-distributed-environment","title":"Unified Distributed Environment","date":"2022-05-14","arxiv_id":"2205.06946","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-computational","slug":"deep-reinforcement-learning-for-computational","title":"Deep Reinforcement Learning for Computational Fluid Dynamics on HPC Systems","date":"2022-05-13","arxiv_id":"2205.06502","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-transmission-control-for-wireless","slug":"distributed-transmission-control-for-wireless","title":"Distributed Transmission Control for Wireless Networks using Multi-Agent Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06800","repositories_listed":1,"syntology":null},{"url":"/paper/modularity-in-neat-reinforcement-learning","slug":"modularity-in-neat-reinforcement-learning","title":"Towards Understanding the Link Between Modularity and Performance in Neural Networks for Reinforcement Learning","date":"2022-05-13","arxiv_id":"2205.06451","repositories_listed":1,"syntology":null}],"record_sha256":"eaea795d42599d4c590c2fe43abe942bdb8a84c1982bb1ad3bd446b4f3c20491","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}