{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/26","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":26,"pages_in_order":152,"rows_per_page":100,"rows":[2501,2600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/25","next":"/task/reinforcement-learning-1/papers/27","papers":[{"url":"/paper/lifelong-inverse-reinforcement-learning-1","slug":"lifelong-inverse-reinforcement-learning-1","title":"Lifelong Inverse Reinforcement Learning","date":"2022-07-01","arxiv_id":"2207.00461","repositories_listed":1,"syntology":null},{"url":"/paper/modular-lifelong-reinforcement-learning-via-1","slug":"modular-lifelong-reinforcement-learning-via-1","title":"Modular Lifelong Reinforcement Learning via Neural Composition","date":"2022-07-01","arxiv_id":"2207.00429","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-lifelong-reinforcement-learning-via-1#ran","syntology_url":"https://syntology.ai/paper/2207.00429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00429"}},"official":{"repos":["lifelong-ml/mendez2022modularlifelongrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","repositories_listed":1,"syntology":null},{"url":"/paper/denoised-mdps-learning-world-models-better","slug":"denoised-mdps-learning-world-models-better","title":"Denoised MDPs: Learning World Models Better Than the World Itself","date":"2022-06-30","arxiv_id":"2206.15477","repositories_listed":1,"syntology":null},{"url":"/paper/mastering-the-game-of-stratego-with-model","slug":"mastering-the-game-of-stratego-with-model","title":"Mastering the Game of Stratego with Model-Free Multiagent Reinforcement Learning","date":"2022-06-30","arxiv_id":"2206.15378","repositories_listed":1,"syntology":null},{"url":"/paper/conditionally-elicitable-dynamic-risk","slug":"conditionally-elicitable-dynamic-risk","title":"Conditionally Elicitable Dynamic Risk Measures for Deep Reinforcement Learning","date":"2022-06-29","arxiv_id":"2206.14666","repositories_listed":1,"syntology":null},{"url":"/paper/daydreamer-world-models-for-physical-robot","slug":"daydreamer-world-models-for-physical-robot","title":"DayDreamer: World Models for Physical Robot Learning","date":"2022-06-28","arxiv_id":"2206.14176","repositories_listed":1,"syntology":null},{"url":"/paper/distspectrl-distributing-specifications-in","slug":"distspectrl-distributing-specifications-in","title":"DistSPECTRL: Distributing Specifications in Multi-Agent Reinforcement Learning Systems","date":"2022-06-28","arxiv_id":"2206.13754","repositories_listed":1,"syntology":null},{"url":"/paper/short-term-plasticity-neurons-learning-to","slug":"short-term-plasticity-neurons-learning-to","title":"Short-Term Plasticity Neurons Learning to Learn and Forget","date":"2022-06-28","arxiv_id":"2206.14048","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-building-control","slug":"zero-shot-building-control","title":"Low Emission Building Control with Zero-Shot Reinforcement Learning","date":"2022-06-28","arxiv_id":"2206.14191","repositories_listed":1,"syntology":null},{"url":"/paper/distinguishing-learning-rules-with-brain","slug":"distinguishing-learning-rules-with-brain","title":"Distinguishing Learning Rules with Brain Machine Interfaces","date":"2022-06-27","arxiv_id":"2206.13448","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distinguishing-learning-rules-with-brain#ran","syntology_url":"https://syntology.ai/paper/2206.13448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13448"}},"official":{"repos":["jacobfulano/learning-rules-with-bmi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-trust-your-simulator-dynamics-aware","slug":"when-to-trust-your-simulator-dynamics-aware","title":"When to Trust Your Simulator: Dynamics-Aware Hybrid Offline-and-Online Reinforcement Learning","date":"2022-06-27","arxiv_id":"2206.13464","repositories_listed":1,"syntology":null},{"url":"/paper/improving-policy-optimization-with-generalist","slug":"improving-policy-optimization-with-generalist","title":"Improving Policy Optimization with Generalist-Specialist Learning","date":"2022-06-26","arxiv_id":"2206.12984","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-asymmetric-and-circular-sequential","slug":"tackling-asymmetric-and-circular-sequential","title":"Tackling Asymmetric and Circular Sequential Social Dilemmas with Reinforcement Learning and Graph-based Tit-for-Tat","date":"2022-06-26","arxiv_id":"2206.12909","repositories_listed":1,"syntology":null},{"url":"/paper/guided-exploration-in-reinforcement-learning","slug":"guided-exploration-in-reinforcement-learning","title":"Guided Exploration in Reinforcement Learning via Monte Carlo Critic Optimization","date":"2022-06-25","arxiv_id":"2206.12674","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-adaptive-1","slug":"reinforcement-learning-based-adaptive-1","title":"Reinforcement learning based adaptive metaheuristics","date":"2022-06-24","arxiv_id":"2206.12233","repositories_listed":1,"syntology":null},{"url":"/paper/cgar-critic-guided-action-redistribution-in","slug":"cgar-critic-guided-action-redistribution-in","title":"CGAR: Critic Guided Action Redistribution in Reinforcement Leaning","date":"2022-06-23","arxiv_id":"2206.11494","repositories_listed":1,"syntology":null},{"url":"/paper/learning-representations-for-control-with","slug":"learning-representations-for-control-with","title":"Multi-Horizon Representations with Hierarchical Forward Models for Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11396","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-car-parking-using-reinforcement","slug":"multi-agent-car-parking-using-reinforcement","title":"Multi-Agent Car Parking using Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.13338","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-linear-support-and-successor","slug":"optimistic-linear-support-and-successor","title":"Optimistic Linear Support and Successor Features as a Basis for Optimal Policy Transfer","date":"2022-06-22","arxiv_id":"2206.11326","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-linear-support-and-successor#ran","syntology_url":"https://syntology.ai/paper/2206.11326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11326"}},"official":{"repos":["lucasalegre/sfols"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pac-assisted-value-factorisation-with","slug":"pac-assisted-value-factorisation-with","title":"PAC: Assisted Value Factorisation with Counterfactual Predictions in Multi-Agent Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11420","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pac-assisted-value-factorisation-with#ran","syntology_url":"https://syntology.ai/paper/2206.11420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11420"}},"official":{"repos":["hanhananderson/pac-marl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-reinforcement-learning-with-finite","slug":"meta-reinforcement-learning-with-finite","title":"Meta Reinforcement Learning with Finite Training Tasks -- a Density Estimation Approach","date":"2022-06-21","arxiv_id":"2206.10716","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-through-1","slug":"robust-deep-reinforcement-learning-through-1","title":"Robust Deep Reinforcement Learning through Bootstrapped Opportunistic Curriculum","date":"2022-06-21","arxiv_id":"2206.10057","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-1#ran","syntology_url":"https://syntology.ai/paper/2206.10057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10057"}},"official":{"repos":["jlwu002/bcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-task-representations-for-offline-meta","slug":"robust-task-representations-for-offline-meta","title":"Robust Task Representations for Offline Meta-Reinforcement Learning via Contrastive Learning","date":"2022-06-21","arxiv_id":"2206.10442","repositories_listed":1,"syntology":null},{"url":"/paper/dna-proximal-policy-optimization-with-a-dual","slug":"dna-proximal-policy-optimization-with-a-dual","title":"DNA: Proximal Policy Optimization with a Dual Network Architecture","date":"2022-06-20","arxiv_id":"2206.10027","repositories_listed":1,"syntology":null},{"url":"/paper/eager-asking-and-answering-questions-for","slug":"eager-asking-and-answering-questions-for","title":"EAGER: Asking and Answering Questions for Automatic Reward Shaping in Language-guided RL","date":"2022-06-20","arxiv_id":"2206.09674","repositories_listed":1,"syntology":null},{"url":"/paper/maser-multi-agent-reinforcement-learning-with","slug":"maser-multi-agent-reinforcement-learning-with","title":"MASER: Multi-Agent Reinforcement Learning with Subgoals Generated from Experience Replay Buffer","date":"2022-06-20","arxiv_id":"2206.10607","repositories_listed":1,"syntology":null},{"url":"/paper/sampling-efficient-deep-reinforcement","slug":"sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","repositories_listed":1,"syntology":null},{"url":"/paper/deep-inverse-reinforcement-learning-for-route","slug":"deep-inverse-reinforcement-learning-for-route","title":"A deep inverse reinforcement learning approach to route choice modeling with context-dependent rewards","date":"2022-06-18","arxiv_id":"2206.10598","repositories_listed":1,"syntology":null},{"url":"/paper/fast-population-based-reinforcement-learning","slug":"fast-population-based-reinforcement-learning","title":"Fast Population-Based Reinforcement Learning on a Single Machine","date":"2022-06-17","arxiv_id":"2206.08888","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-population-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.08888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08888"}},"official":null}},{"url":"/paper/logic-based-reward-shaping-for-multi-agent","slug":"logic-based-reward-shaping-for-multi-agent","title":"Logic-based Reward Shaping for Multi-Agent Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08881","repositories_listed":1,"syntology":null},{"url":"/paper/saferl-kit-evaluating-efficient-reinforcement","slug":"saferl-kit-evaluating-efficient-reinforcement","title":"SafeRL-Kit: Evaluating Efficient Reinforcement Learning Methods for Safe Autonomous Driving","date":"2022-06-17","arxiv_id":"2206.08528","repositories_listed":1,"syntology":null},{"url":"/paper/smpl-simulated-industrial-manufacturing-and","slug":"smpl-simulated-industrial-manufacturing-and","title":"SMPL: Simulated Industrial Manufacturing and Process Control Learning Environments","date":"2022-06-17","arxiv_id":"2206.08851","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smpl-simulated-industrial-manufacturing-and#ran","syntology_url":"https://syntology.ai/paper/2206.08851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08851"}},"official":{"repos":["smpl-env/smpl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-state-of-sparse-training-in-deep","slug":"the-state-of-sparse-training-in-deep","title":"The State of Sparse Training in Deep Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.10369","repositories_listed":1,"syntology":null},{"url":"/paper/towards-human-level-bimanual-dexterous","slug":"towards-human-level-bimanual-dexterous","title":"Towards Human-Level Bimanual Dexterous Manipulation with Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08686","repositories_listed":1,"syntology":null},{"url":"/paper/barrier-certified-safety-learning-control","slug":"barrier-certified-safety-learning-control","title":"Barrier Certified Safety Learning Control: When Sum-of-Square Programming Meets Reinforcement Learning","date":"2022-06-16","arxiv_id":"2206.07915","repositories_listed":1,"syntology":null},{"url":"/paper/double-check-your-state-before-trusting-it","slug":"double-check-your-state-before-trusting-it","title":"Double Check Your State Before Trusting It: Confidence-Aware Bidirectional Offline Model-Based Imagination","date":"2022-06-16","arxiv_id":"2206.07989","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/double-check-your-state-before-trusting-it#ran","syntology_url":"https://syntology.ai/paper/2206.07989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07989"}},"official":{"repos":["dmksjfl/CABI"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/reinforcement-learning-enhanced-shared","slug":"reinforcement-learning-enhanced-shared","title":"Reinforcement Learning-enhanced Shared-account Cross-domain Sequential Recommendation","date":"2022-06-16","arxiv_id":"2206.08088","repositories_listed":1,"syntology":null},{"url":"/paper/search-based-testing-approach-for-deep","slug":"search-based-testing-approach-for-deep","title":"A Search-Based Testing Approach for Deep Reinforcement Learning Agents","date":"2022-06-15","arxiv_id":"2206.07813","repositories_listed":1,"syntology":null},{"url":"/paper/training-discrete-deep-generative-models-via","slug":"training-discrete-deep-generative-models-via","title":"Training Discrete Deep Generative Models via Gapped Straight-Through Estimator","date":"2022-06-15","arxiv_id":"2206.07235","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-discrete-deep-generative-models-via#ran","syntology_url":"https://syntology.ai/paper/2206.07235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07235"}},"official":{"repos":["chijames/gst"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/defending-observation-attacks-in-deep","slug":"defending-observation-attacks-in-deep","title":"Defending Observation Attacks in Deep Reinforcement Learning via Detection and Denoising","date":"2022-06-14","arxiv_id":"2206.07188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/defending-observation-attacks-in-deep#ran","syntology_url":"https://syntology.ai/paper/2206.07188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07188"}},"official":{"repos":["ZikangXiong/rl-detect-and-denoise-defense"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rosgas-adaptive-social-bot-detection-with","slug":"rosgas-adaptive-social-bot-detection-with","title":"RoSGAS: Adaptive Social Bot Detection with Reinforced Self-Supervised GNN Architecture Search","date":"2022-06-14","arxiv_id":"2206.06757","repositories_listed":1,"syntology":null},{"url":"/paper/transformers-are-meta-reinforcement-learners-1","slug":"transformers-are-meta-reinforcement-learners-1","title":"Transformers are Meta-Reinforcement Learners","date":"2022-06-14","arxiv_id":"2206.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transformers-are-meta-reinforcement-learners-1#ran","syntology_url":"https://syntology.ai/paper/2206.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06614"}},"official":{"repos":["luckeciano/transformers-metarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/universally-expressive-communication-in-multi","slug":"universally-expressive-communication-in-multi","title":"Universally Expressive Communication in Multi-Agent Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.06758","repositories_listed":1,"syntology":null},{"url":"/paper/ign-implicit-generative-networks","slug":"ign-implicit-generative-networks","title":"IGN : Implicit Generative Networks","date":"2022-06-13","arxiv_id":"2206.05860","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-placement-of","slug":"reinforcement-learning-based-placement-of","title":"Reinforcement Learning-based Placement of Charging Stations in Urban Road Networks","date":"2022-06-13","arxiv_id":"2206.06011","repositories_listed":1,"syntology":null},{"url":"/paper/case-based-inverse-reinforcement-learning","slug":"case-based-inverse-reinforcement-learning","title":"Case-Based Inverse Reinforcement Learning Using Temporal Coherence","date":"2022-06-12","arxiv_id":"2206.05827","repositories_listed":1,"syntology":null},{"url":"/paper/roi-constrained-bidding-via-curriculum-guided","slug":"roi-constrained-bidding-via-curriculum-guided","title":"ROI-Constrained Bidding via Curriculum-Guided Bayesian Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05240","repositories_listed":1,"syntology":null},{"url":"/paper/a-relational-intervention-approach-for-1","slug":"a-relational-intervention-approach-for-1","title":"A Relational Intervention Approach for Unsupervised Dynamics Generalization in Model-Based Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04551","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-reinforcement-learning-via-2","slug":"towards-safe-reinforcement-learning-via-2","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value-at-Risk","date":"2022-06-09","arxiv_id":"2206.04436","repositories_listed":1,"syntology":null},{"url":"/paper/value-memory-graph-a-graph-structured-world","slug":"value-memory-graph-a-graph-structured-world","title":"Value Memory Graph: A Graph-Structured World Model for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04384","repositories_listed":1,"syntology":null},{"url":"/paper/designing-reinforcement-learning-algorithms","slug":"designing-reinforcement-learning-algorithms","title":"Designing Reinforcement Learning Algorithms for Digital Interventions: Pre-implementation Guidelines","date":"2022-06-08","arxiv_id":"2206.03944","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-voltage-in-power-distribution","slug":"stabilizing-voltage-in-power-distribution","title":"Stabilizing Voltage in Power Distribution Networks via Multi-Agent Reinforcement Learning with Transformer","date":"2022-06-08","arxiv_id":"2206.03721","repositories_listed":1,"syntology":null},{"url":"/paper/deeptpi-test-point-insertion-with-deep","slug":"deeptpi-test-point-insertion-with-deep","title":"DeepTPI: Test Point Insertion with Deep Reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.06975","repositories_listed":1,"syntology":null},{"url":"/paper/discrete-state-action-abstraction-via-the","slug":"discrete-state-action-abstraction-via-the","title":"Discrete State-Action Abstraction via the Successor Representation","date":"2022-06-07","arxiv_id":"2206.03467","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-i-ll-go-offline-goal-conditioned","slug":"how-far-i-ll-go-offline-goal-conditioned","title":"How Far I'll Go: Offline Goal-Conditioned Reinforcement Learning via $f$-Advantage Regression","date":"2022-06-07","arxiv_id":"2206.03023","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-far-i-ll-go-offline-goal-conditioned#ran","syntology_url":"https://syntology.ai/paper/2206.03023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03023"}},"official":{"repos":["jasonma2016/gofar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/look-back-when-surprised-stabilizing-reverse","slug":"look-back-when-surprised-stabilizing-reverse","title":"Introspective Experience Replay: Look Back When Surprised","date":"2022-06-07","arxiv_id":"2206.03171","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-safe-exploration-using-safety-state","slug":"enhancing-safe-exploration-using-safety-state","title":"Effects of Safety State Augmentation on Safe Exploration","date":"2022-06-06","arxiv_id":"2206.02675","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-nav-a-library-for-neurally-plausible","slug":"neuro-nav-a-library-for-neurally-plausible","title":"Neuro-Nav: A Library for Neurally-Plausible Reinforcement Learning","date":"2022-06-06","arxiv_id":"2206.03312","repositories_listed":1,"syntology":null},{"url":"/paper/rorl-robust-offline-reinforcement-learning","slug":"rorl-robust-offline-reinforcement-learning","title":"RORL: Robust Offline Reinforcement Learning via Conservative Smoothing","date":"2022-06-06","arxiv_id":"2206.02829","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rorl-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.02829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02829"}},"official":{"repos":["yangrui2015/rorl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-tabula-rasa-reincarnating","slug":"beyond-tabula-rasa-reincarnating","title":"Reincarnating Reinforcement Learning: Reusing Prior Computation to Accelerate Progress","date":"2022-06-03","arxiv_id":"2206.01626","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-tabula-rasa-reincarnating#ran","syntology_url":"https://syntology.ai/paper/2206.01626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01626"}},"official":{"repos":["google-research/reincarnating_rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-transformer-q-networks-for-partially","slug":"deep-transformer-q-networks-for-partially","title":"Deep Transformer Q-Networks for Partially Observable Reinforcement Learning","date":"2022-06-02","arxiv_id":"2206.01078","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-transformer-q-networks-for-partially#ran","syntology_url":"https://syntology.ai/paper/2206.01078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01078"}},"official":{"repos":["kevslinger/dtqn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neuralsympcheck-a-symptom-checking-and","slug":"neuralsympcheck-a-symptom-checking-and","title":"NeuralSympCheck: A Symptom Checking and Disease Diagnostic Neural Model with Logic Regularization","date":"2022-06-02","arxiv_id":"2206.00906","repositories_listed":1,"syntology":null},{"url":"/paper/when-does-return-conditioned-supervised","slug":"when-does-return-conditioned-supervised","title":"When does return-conditioned supervised learning work for offline reinforcement learning?","date":"2022-06-02","arxiv_id":"2206.01079","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/when-does-return-conditioned-supervised#ran","syntology_url":"https://syntology.ai/paper/2206.01079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01079"}},"official":{"repos":["davidbrandfonbrener/rcsl-paper"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dm-2-distributed-multi-agent-reinforcement","slug":"dm-2-distributed-multi-agent-reinforcement","title":"DM$^2$: Decentralized Multi-Agent Reinforcement Learning for Distribution Matching","date":"2022-06-01","arxiv_id":"2206.00233","repositories_listed":1,"syntology":null},{"url":"/paper/neural-improvement-heuristics-for-preference","slug":"neural-improvement-heuristics-for-preference","title":"Neural Improvement Heuristics for Graph Combinatorial Optimization Problems","date":"2022-06-01","arxiv_id":"2206.00383","repositories_listed":1,"syntology":null},{"url":"/paper/resact-reinforcing-long-term-engagement-in","slug":"resact-reinforcing-long-term-engagement-in","title":"ResAct: Reinforcing Long-term Engagement in Sequential Recommendation with Residual Actor","date":"2022-06-01","arxiv_id":"2206.02620","repositories_listed":1,"syntology":null},{"url":"/paper/a-meta-reinforcement-learning-approach-for","slug":"a-meta-reinforcement-learning-approach-for","title":"A Meta Reinforcement Learning Approach for Predictive Autoscaling in the Cloud","date":"2022-05-31","arxiv_id":"2205.15795","repositories_listed":1,"syntology":null},{"url":"/paper/graph-backup-data-efficient-backup-exploiting","slug":"graph-backup-data-efficient-backup-exploiting","title":"Graph Backup: Data Efficient Backup Exploiting Markovian Transitions","date":"2022-05-31","arxiv_id":"2205.15824","repositories_listed":1,"syntology":null},{"url":"/paper/human-ai-shared-control-via-frequency-based","slug":"human-ai-shared-control-via-frequency-based","title":"Human-AI Shared Control via Policy Dissection","date":"2022-05-31","arxiv_id":"2206.00152","repositories_listed":1,"syntology":null},{"url":"/paper/iglu-gridworld-simple-and-fast-environment","slug":"iglu-gridworld-simple-and-fast-environment","title":"IGLU Gridworld: Simple and Fast Environment for Embodied Dialog Agents","date":"2022-05-31","arxiv_id":"2206.00142","repositories_listed":1,"syntology":null},{"url":"/paper/dep-rl-embodied-exploration-for-reinforcement","slug":"dep-rl-embodied-exploration-for-reinforcement","title":"DEP-RL: Embodied Exploration for Reinforcement Learning in Overactuated and Musculoskeletal Systems","date":"2022-05-30","arxiv_id":"2206.00484","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dep-rl-embodied-exploration-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.00484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00484"}},"official":{"repos":["martius-lab/depRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-reward-poisoning-attacks-on-online#ran","syntology_url":"https://syntology.ai/paper/2205.14842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14842"}},"official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-is-a","slug":"multi-agent-reinforcement-learning-is-a","title":"Multi-Agent Reinforcement Learning is a Sequence Modeling Problem","date":"2022-05-30","arxiv_id":"2205.14953","repositories_listed":1,"syntology":null},{"url":"/paper/non-markovian-reward-modelling-from","slug":"non-markovian-reward-modelling-from","title":"Non-Markovian Reward Modelling from Trajectory Labels via Interpretable Multiple Instance Learning","date":"2022-05-30","arxiv_id":"2205.15367","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-terminator","slug":"reinforcement-learning-with-a-terminator","title":"Reinforcement Learning with a Terminator","date":"2022-05-30","arxiv_id":"2205.15376","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-a-terminator#ran","syntology_url":"https://syntology.ai/paper/2205.15376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15376"}},"official":{"repos":["guytenn/terminator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlx2-training-a-sparse-deep-reinforcement","slug":"rlx2-training-a-sparse-deep-reinforcement","title":"RLx2: Training a Sparse Deep Reinforcement Learning Model from Scratch","date":"2022-05-30","arxiv_id":"2205.15043","repositories_listed":1,"syntology":null},{"url":"/paper/learning-security-strategies-through-game","slug":"learning-security-strategies-through-game","title":"Learning Security Strategies through Game Play and Optimal Stopping","date":"2022-05-29","arxiv_id":"2205.14694","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-robustness-of-safe-reinforcement","slug":"on-the-robustness-of-safe-reinforcement","title":"On the Robustness of Safe Reinforcement Learning under Observational Perturbations","date":"2022-05-29","arxiv_id":"2205.14691","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-the-robustness-of-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14691"}},"official":{"repos":["liuzuxin/safe-rl-robustness"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-benefits-of-representational","slug":"provable-benefits-of-representational","title":"Provable Benefits of Representational Transfer in Reinforcement Learning","date":"2022-05-29","arxiv_id":"2205.14571","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-branch-and-bound","slug":"reinforcement-learning-for-branch-and-bound","title":"Reinforcement Learning for Branch-and-Bound Optimisation using Retrospective Trajectories","date":"2022-05-28","arxiv_id":"2205.14345","repositories_listed":1,"syntology":null},{"url":"/paper/fedformer-contextual-federation-with","slug":"fedformer-contextual-federation-with","title":"FedFormer: Contextual Federation with Attention in Reinforcement Learning","date":"2022-05-27","arxiv_id":"2205.13697","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedformer-contextual-federation-with#ran","syntology_url":"https://syntology.ai/paper/2205.13697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13697"}},"official":{"repos":["liamhebert/FedFormer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iglu-2022-interactive-grounded-language","slug":"iglu-2022-interactive-grounded-language","title":"IGLU 2022: Interactive Grounded Language Understanding in a Collaborative Environment at NeurIPS 2022","date":"2022-05-27","arxiv_id":"2205.13771","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-solve-combinatorial-graph","slug":"learning-to-solve-combinatorial-graph","title":"Learning to Solve Combinatorial Graph Partitioning Problems via Efficient Exploration","date":"2022-05-27","arxiv_id":"2205.14105","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-solve-combinatorial-graph#ran","syntology_url":"https://syntology.ai/paper/2205.14105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14105"}},"official":{"repos":["tomdbar/ecord"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/drlcomplex-reconstruction-of-protein","slug":"drlcomplex-reconstruction-of-protein","title":"DRLComplex: Reconstruction of protein quaternary structures using deep reinforcement learning","date":"2022-05-26","arxiv_id":"2205.13594","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-network-reconfiguration-for-entropy","slug":"dynamic-network-reconfiguration-for-entropy","title":"Dynamic Network Reconfiguration for Entropy Maximization using Deep Reinforcement Learning","date":"2022-05-26","arxiv_id":"2205.13578","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-approach-for-mapping","slug":"reinforcement-learning-approach-for-mapping","title":"Reinforcement Learning Approach for Mapping Applications to Dataflow-Based Coarse-Grained Reconfigurable Array","date":"2022-05-26","arxiv_id":"2205.13675","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-reinforcement-adaptation-for-1","slug":"unsupervised-reinforcement-adaptation-for-1","title":"Unsupervised Reinforcement Adaptation for Class-Imbalanced Text Classification","date":"2022-05-26","arxiv_id":"2205.13139","repositories_listed":1,"syntology":null},{"url":"/paper/impartial-games-a-challenge-for-reinforcement","slug":"impartial-games-a-challenge-for-reinforcement","title":"Impartial Games: A Challenge for Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12787","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-query-internet-text-for-informing","slug":"learning-to-query-internet-text-for-informing","title":"Learning to Query Internet Text for Informing Reinforcement Learning Agents","date":"2022-05-25","arxiv_id":"2205.13079","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-knowledge-alignment-with","slug":"multimodal-knowledge-alignment-with","title":"Multimodal Knowledge Alignment with Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12630","repositories_listed":1,"syntology":null},{"url":"/paper/rlprompt-optimizing-discrete-text-prompts","slug":"rlprompt-optimizing-discrete-text-prompts","title":"RLPrompt: Optimizing Discrete Text Prompts with Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12548","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/rlprompt-optimizing-discrete-text-prompts#ran","syntology_url":"https://syntology.ai/paper/2205.12548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12548"}},"official":{"repos":["mingkaid/rl-prompt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/scalable-multi-agent-model-based","slug":"scalable-multi-agent-model-based","title":"Scalable Multi-Agent Model-Based Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.15023","repositories_listed":1,"syntology":null},{"url":"/paper/skill-machines-temporal-logic-composition-in","slug":"skill-machines-temporal-logic-composition-in","title":"Skill Machines: Temporal Logic Skill Composition in Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12532","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/skill-machines-temporal-logic-composition-in#ran","syntology_url":"https://syntology.ai/paper/2205.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12532"}},"official":{"repos":["geraudnt/skill_machines"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/tiered-reinforcement-learning-pessimism-in","slug":"tiered-reinforcement-learning-pessimism-in","title":"Tiered Reinforcement Learning: Pessimism in the Face of Uncertainty and Constant Regret","date":"2022-05-25","arxiv_id":"2205.12418","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/tiered-reinforcement-learning-pessimism-in#ran","syntology_url":"https://syntology.ai/paper/2205.12418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12418"}},"official":{"repos":["jiaweihhuang/tiered-rl-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/concurrent-credit-assignment-for-data","slug":"concurrent-credit-assignment-for-data","title":"Concurrent Credit Assignment for Data-efficient Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12020","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-class","slug":"deep-reinforcement-learning-for-multi-class","title":"Deep Reinforcement Learning for Multi-class Imbalanced Training","date":"2022-05-24","arxiv_id":"2205.12070","repositories_listed":1,"syntology":null},{"url":"/paper/meta-policy-learning-for-cold-start","slug":"meta-policy-learning-for-cold-start","title":"Meta Policy Learning for Cold-Start Conversational Recommendation","date":"2022-05-24","arxiv_id":"2205.11788","repositories_listed":1,"syntology":null},{"url":"/paper/an-evaluation-study-of-intrinsic-motivation","slug":"an-evaluation-study-of-intrinsic-motivation","title":"An Evaluation Study of Intrinsic Motivation Techniques applied to Reinforcement Learning over Hard Exploration Environments","date":"2022-05-23","arxiv_id":"2205.11184","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-evaluation-study-of-intrinsic-motivation#ran","syntology_url":"https://syntology.ai/paper/2205.11184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11184"}},"official":{"repos":["aklein1995/intrinsic_motivation_techniques_study"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"b7aa922f0b0118df9db535aac7306e680c3cbac917dd593a96694a10d853c8d6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}