{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/24","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":24,"pages_in_order":152,"rows_per_page":100,"rows":[2301,2400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/23","next":"/task/reinforcement-learning-1/papers/25","papers":[{"url":"/paper/on-the-feasibility-of-cross-task-transfer","slug":"on-the-feasibility-of-cross-task-transfer","title":"On the Feasibility of Cross-Task Transfer with Model-Based Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10763","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/on-the-feasibility-of-cross-task-transfer#ran","syntology_url":"https://syntology.ai/paper/2210.10763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10763"}},"official":{"repos":["mlpc-ucsd/xtra"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-offline-reinforcement-learning-with","slug":"robust-offline-reinforcement-learning-with","title":"Robust Offline Reinforcement Learning with Gradient Penalty and Constraint Relaxation","date":"2022-10-19","arxiv_id":"2210.10469","repositories_listed":1,"syntology":null},{"url":"/paper/when-to-ask-for-help-proactive-interventions","slug":"when-to-ask-for-help-proactive-interventions","title":"When to Ask for Help: Proactive Interventions in Autonomous Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10765","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/when-to-ask-for-help-proactive-interventions#ran","syntology_url":"https://syntology.ai/paper/2210.10765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10765"}},"official":{"repos":["tajwarfahim/proactive_interventions"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ceip-combining-explicit-and-implicit-priors","slug":"ceip-combining-explicit-and-implicit-priors","title":"CEIP: Combining Explicit and Implicit Priors for Reinforcement Learning with Demonstrations","date":"2022-10-18","arxiv_id":"2210.09496","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ceip-combining-explicit-and-implicit-priors#ran","syntology_url":"https://syntology.ai/paper/2210.09496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09496"}},"official":{"repos":["289371298/ceip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/curriculum-reinforcement-learning-using","slug":"curriculum-reinforcement-learning-using","title":"Curriculum Reinforcement Learning using Optimal Transport via Gradual Domain Adaptation","date":"2022-10-18","arxiv_id":"2210.10195","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/curriculum-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2210.10195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10195"}},"official":{"repos":["peidehuang/gradient"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-black-box-reinforcement-learning-with","slug":"deep-black-box-reinforcement-learning-with","title":"Deep Black-Box Reinforcement Learning with Movement Primitives","date":"2022-10-18","arxiv_id":"2210.09622","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-black-box-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2210.09622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09622"}},"official":{"repos":["ALRhub/fancy_gym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-value-function-learning-for","slug":"rethinking-value-function-learning-for","title":"Rethinking Value Function Learning for Generalization in Reinforcement Learning","date":"2022-10-18","arxiv_id":"2210.09960","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/rethinking-value-function-learning-for#ran","syntology_url":"https://syntology.ai/paper/2210.09960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09960"}},"official":{"repos":["snu-mllab/dcpg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-generative-user-simulator-with-gpt-based","slug":"a-generative-user-simulator-with-gpt-based","title":"A Generative User Simulator with GPT-based Architecture and Goal State Tracking for Reinforced Multi-Domain Dialog Systems","date":"2022-10-17","arxiv_id":"2210.08692","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-generative-user-simulator-with-gpt-based#ran","syntology_url":"https://syntology.ai/paper/2210.08692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08692"}},"official":{"repos":["thu-spmi/gus"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-uncertainty-in-deep-state-space-models-for","slug":"on-uncertainty-in-deep-state-space-models-for","title":"On Uncertainty in Deep State Space Models for Model-Based Reinforcement Learning","date":"2022-10-17","arxiv_id":"2210.09256","repositories_listed":1,"syntology":null},{"url":"/paper/teacher-forcing-recovers-reward-functions-for","slug":"teacher-forcing-recovers-reward-functions-for","title":"Teacher Forcing Recovers Reward Functions for Text Generation","date":"2022-10-17","arxiv_id":"2210.08708","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teacher-forcing-recovers-reward-functions-for#ran","syntology_url":"https://syntology.ai/paper/2210.08708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08708"}},"official":{"repos":["manga-uofa/lmreward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-policy-guided-imitation-approach-for","slug":"a-policy-guided-imitation-approach-for","title":"A Policy-Guided Imitation Approach for Offline Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08323","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-policy-guided-imitation-approach-for#ran","syntology_url":"https://syntology.ai/paper/2210.08323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08323"}},"official":{"repos":["ryanxhr/por"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-the-robustness-of-pecnet","slug":"analyzing-the-robustness-of-pecnet","title":"G-PECNet: Towards a Generalizable Pedestrian Trajectory Prediction System","date":"2022-10-15","arxiv_id":"2210.09846","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/analyzing-the-robustness-of-pecnet#ran","syntology_url":"https://syntology.ai/paper/2210.09846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09846"}},"official":{"repos":["aryan-garg/pecnet-pedestrian-trajectory-prediction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-update-your-model-constrained-model","slug":"when-to-update-your-model-constrained-model","title":"When to Update Your Model: Constrained Model-based Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08349","repositories_listed":1,"syntology":null},{"url":"/paper/abstract-to-executable-trajectory-translation","slug":"abstract-to-executable-trajectory-translation","title":"Abstract-to-Executable Trajectory Translation for One-Shot Task Generalization","date":"2022-10-14","arxiv_id":"2210.07658","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-reward-estimation-for","slug":"distributional-reward-estimation-for","title":"Distributional Reward Estimation for Effective Multi-Agent Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07636","repositories_listed":1,"syntology":null},{"url":"/paper/frame-mining-a-free-lunch-for-learning","slug":"frame-mining-a-free-lunch-for-learning","title":"Frame Mining: a Free Lunch for Learning Robotic Manipulation from 3D Point Clouds","date":"2022-10-14","arxiv_id":"2210.07442","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/frame-mining-a-free-lunch-for-learning#ran","syntology_url":"https://syntology.ai/paper/2210.07442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07442"}},"official":{"repos":["xuanlinli17/corl_22_frame_mining"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/just-round-quantized-observation-spaces","slug":"just-round-quantized-observation-spaces","title":"Just Round: Quantized Observation Spaces Enable Memory Efficient Learning of Dynamic Locomotion","date":"2022-10-14","arxiv_id":"2210.08065","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-safe-deep-reinforcement-learning","slug":"model-based-safe-deep-reinforcement-learning","title":"Model-based Safe Deep Reinforcement Learning via a Constrained Proximal Policy Optimization Algorithm","date":"2022-10-14","arxiv_id":"2210.07573","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/model-based-safe-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.07573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07573"}},"official":{"repos":["akjayant/mbppol"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-model-based-reinforcement-learning-with-2","slug":"safe-model-based-reinforcement-learning-with-2","title":"Safe Model-Based Reinforcement Learning with an Uncertainty-Aware Reachability Certificate","date":"2022-10-14","arxiv_id":"2210.07553","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-model-based-reinforcement-learning-with-2#ran","syntology_url":"https://syntology.ai/paper/2210.07553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07553"}},"official":{"repos":["ManUtdMoon/Safe_MBRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/skill-based-reinforcement-learning-with","slug":"skill-based-reinforcement-learning-with","title":"Skill-Based Reinforcement Learning with Intrinsic Reward Matching","date":"2022-10-14","arxiv_id":"2210.07426","repositories_listed":1,"syntology":null},{"url":"/paper/touplegdd-a-fine-designed-solution-of","slug":"touplegdd-a-fine-designed-solution-of","title":"ToupleGDD: A Fine-Designed Solution of Influence Maximization by Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07500","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/touplegdd-a-fine-designed-solution-of#ran","syntology_url":"https://syntology.ai/paper/2210.07500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07500"}},"official":{"repos":["Dtrycode/ToupleGDD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wild-scav-benchmarking-fps-gaming-ai-on","slug":"wild-scav-benchmarking-fps-gaming-ai-on","title":"WILD-SCAV: Benchmarking FPS Gaming AI on Unity3D-based Environments","date":"2022-10-14","arxiv_id":"2210.09026","repositories_listed":1,"syntology":null},{"url":"/paper/a-mixture-of-surprises-for-unsupervised","slug":"a-mixture-of-surprises-for-unsupervised","title":"A Mixture of Surprises for Unsupervised Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.06702","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-mixture-of-surprises-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2210.06702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06702"}},"official":{"repos":["leaplabthu/moss"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstrap-advantage-estimation-for-policy","slug":"bootstrap-advantage-estimation-for-policy","title":"Bootstrap Advantage Estimation for Policy Optimization in Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.07312","repositories_listed":1,"syntology":null},{"url":"/paper/harfang3d-dog-fight-sandbox-a-reinforcement","slug":"harfang3d-dog-fight-sandbox-a-reinforcement","title":"Harfang3D Dog-Fight Sandbox: A Reinforcement Learning Research Platform for the Customized Control Tasks of Fighter Aircrafts","date":"2022-10-13","arxiv_id":"2210.07282","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-dynamic-algorithm-configuration","slug":"multi-agent-dynamic-algorithm-configuration","title":"Multi-agent Dynamic Algorithm Configuration","date":"2022-10-13","arxiv_id":"2210.06835","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-dynamic-algorithm-configuration#ran","syntology_url":"https://syntology.ai/paper/2210.06835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06835"}},"official":{"repos":["lamda-bbo/madac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sustainable-online-reinforcement-learning-for","slug":"sustainable-online-reinforcement-learning-for","title":"Sustainable Online Reinforcement Learning for Auto-bidding","date":"2022-10-13","arxiv_id":"2210.07006","repositories_listed":1,"syntology":null},{"url":"/paper/towards-trustworthy-automatic-diagnosis","slug":"towards-trustworthy-automatic-diagnosis","title":"Towards Trustworthy Automatic Diagnosis Systems by Emulating Doctors' Reasoning with Deep Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.07198","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-trustworthy-automatic-diagnosis#ran","syntology_url":"https://syntology.ai/paper/2210.07198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07198"}},"official":{"repos":["mila-iqia/casande-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-reinforcement-learning-with-self","slug":"visual-reinforcement-learning-with-self","title":"Visual Reinforcement Learning with Self-Supervised 3D Representations","date":"2022-10-13","arxiv_id":"2210.07241","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-reinforcement-learning-with-self#ran","syntology_url":"https://syntology.ai/paper/2210.07241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07241"}},"official":{"repos":["YanjieZe/rl3d"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-alternating-offline","slug":"a-unified-framework-for-alternating-offline","title":"A Unified Framework for Alternating Offline Model Training and Policy Learning","date":"2022-10-12","arxiv_id":"2210.05922","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-alternating-offline#ran","syntology_url":"https://syntology.ai/paper/2210.05922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05922"}},"official":{"repos":["shentao-yang/ampl_neurips2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/centralized-training-with-hybrid-execution-in","slug":"centralized-training-with-hybrid-execution-in","title":"Centralized Training with Hybrid Execution in Multi-Agent Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.06274","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-retrospection-honing-in-on-1","slug":"contrastive-retrospection-honing-in-on-1","title":"Contrastive Retrospection: honing in on critical steps for rapid learning and generalization in RL","date":"2022-10-12","arxiv_id":"2210.05845","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-adversarial-training-without","slug":"efficient-adversarial-training-without","title":"Efficient Adversarial Training without Attacking: Worst-Case-Aware Robust Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.05927","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-adversarial-training-without#ran","syntology_url":"https://syntology.ai/paper/2210.05927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05927"}},"official":{"repos":["umd-huang-lab/wocar-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-supervised-offline-reinforcement-1","slug":"semi-supervised-offline-reinforcement-1","title":"Semi-Supervised Offline Reinforcement Learning with Action-Free Trajectories","date":"2022-10-12","arxiv_id":"2210.06518","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/semi-supervised-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2210.06518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06518"}},"official":{"repos":["facebookresearch/ssorl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/conserweightive-behavioral-cloning-for","slug":"conserweightive-behavioral-cloning-for","title":"Reliable Conditioning of Behavioral Cloning for Offline Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05158","repositories_listed":1,"syntology":null},{"url":"/paper/dhrl-a-graph-based-approach-for-long-horizon","slug":"dhrl-a-graph-based-approach-for-long-horizon","title":"DHRL: A Graph-Based Approach for Long-Horizon and Sparse Hierarchical Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05150","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dhrl-a-graph-based-approach-for-long-horizon#ran","syntology_url":"https://syntology.ai/paper/2210.05150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05150"}},"official":null}},{"url":"/paper/discovered-policy-optimisation","slug":"discovered-policy-optimisation","title":"Discovered Policy Optimisation","date":"2022-10-11","arxiv_id":"2210.05639","repositories_listed":1,"syntology":null},{"url":"/paper/marllib-extending-rllib-for-multi-agent","slug":"marllib-extending-rllib-for-multi-agent","title":"MARLlib: A Scalable and Efficient Multi-agent Reinforcement Learning Library","date":"2022-10-11","arxiv_id":"2210.13708","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/marllib-extending-rllib-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2210.13708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13708"}},"official":{"repos":["replicable-marl/marllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mastering-the-game-of-no-press-diplomacy-via","slug":"mastering-the-game-of-no-press-diplomacy-via","title":"Mastering the Game of No-Press Diplomacy via Human-Regularized Reinforcement Learning and Planning","date":"2022-10-11","arxiv_id":"2210.05492","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mastering-the-game-of-no-press-diplomacy-via#ran","syntology_url":"https://syntology.ai/paper/2210.05492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05492"}},"official":null}},{"url":"/paper/multi-object-navigation-with-dynamically","slug":"multi-object-navigation-with-dynamically","title":"Multi-Object Navigation with dynamically learned neural implicit representations","date":"2022-10-11","arxiv_id":"2210.05129","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-object-navigation-with-dynamically#ran","syntology_url":"https://syntology.ai/paper/2210.05129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05129"}},"official":{"repos":["PierreMarza/dynamic_implicit_representations"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comprehensive-survey-of-data-augmentation","slug":"a-comprehensive-survey-of-data-augmentation","title":"A Comprehensive Survey of Data Augmentation in Visual Reinforcement Learning","date":"2022-10-10","arxiv_id":"2210.04561","repositories_listed":1,"syntology":null},{"url":"/paper/a-policy-gradient-approach-for-finite-horizon","slug":"a-policy-gradient-approach-for-finite-horizon","title":"A policy gradient approach for Finite Horizon Constrained Markov Decision Processes","date":"2022-10-10","arxiv_id":"2210.04527","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/a-policy-gradient-approach-for-finite-horizon#ran","syntology_url":"https://syntology.ai/paper/2210.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04527"}},"official":{"repos":["gsoumyajit/Finite-Horizon-with-constraints"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/benchmarking-reinforcement-learning-1","slug":"benchmarking-reinforcement-learning-1","title":"Benchmarking Reinforcement Learning Techniques for Autonomous Navigation","date":"2022-10-10","arxiv_id":"2210.04839","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2210.04839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04839"}},"official":null}},{"url":"/paper/experiential-explanations-for-reinforcement","slug":"experiential-explanations-for-reinforcement","title":"Experiential Explanations for Reinforcement Learning","date":"2022-10-10","arxiv_id":"2210.04723","repositories_listed":1,"syntology":null},{"url":"/paper/in-hand-object-rotation-via-rapid-motor","slug":"in-hand-object-rotation-via-rapid-motor","title":"In-Hand Object Rotation via Rapid Motor Adaptation","date":"2022-10-10","arxiv_id":"2210.04887","repositories_listed":1,"syntology":null},{"url":"/paper/learning-credit-assignment-for-cooperative","slug":"learning-credit-assignment-for-cooperative","title":"Learning Explicit Credit Assignment for Cooperative Multi-Agent Reinforcement Learning via Polarization Policy Gradient","date":"2022-10-10","arxiv_id":"2210.05367","repositories_listed":1,"syntology":null},{"url":"/paper/multiagent-reinforcement-learning-based-on","slug":"multiagent-reinforcement-learning-based-on","title":"Multiagent Reinforcement Learning Based on Fusion-Multiactor-Attention-Critic for Multiple-Unmanned-Aerial-Vehicle Navigation Control","date":"2022-10-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/decomposed-mutual-information-optimization","slug":"decomposed-mutual-information-optimization","title":"Decomposed Mutual Information Optimization for Generalized Context in Meta-Reinforcement Learning","date":"2022-10-09","arxiv_id":"2210.04209","repositories_listed":1,"syntology":null},{"url":"/paper/skeleton2humanoid-animating-simulated","slug":"skeleton2humanoid-animating-simulated","title":"Skeleton2Humanoid: Animating Simulated Characters for Physically-plausible Motion In-betweening","date":"2022-10-09","arxiv_id":"2210.04294","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-bayesian-model-based-value","slug":"conservative-bayesian-model-based-value","title":"Conservative Bayesian Model-Based Value Expansion for Offline Policy Optimization","date":"2022-10-07","arxiv_id":"2210.03802","repositories_listed":1,"syntology":null},{"url":"/paper/flexible-attention-based-multi-policy-fusion-1","slug":"flexible-attention-based-multi-policy-fusion-1","title":"Flexible Attention-Based Multi-Policy Fusion for Efficient Deep Reinforcement Learning","date":"2022-10-07","arxiv_id":"2210.03729","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flexible-attention-based-multi-policy-fusion-1#ran","syntology_url":"https://syntology.ai/paper/2210.03729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03729"}},"official":{"repos":["pascalson/kgrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-your-data-hiding-backdoors-in-offline","slug":"mind-your-data-hiding-backdoors-in-offline","title":"BAFFLE: Hiding Backdoors in Offline Reinforcement Learning Datasets","date":"2022-10-07","arxiv_id":"2210.04688","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-directed-controller-synthesis-via","slug":"scaling-directed-controller-synthesis-via","title":"Exploration Policies for On-the-Fly Controller Synthesis: A Reinforcement Learning Approach","date":"2022-10-07","arxiv_id":"2210.05393","repositories_listed":1,"syntology":null},{"url":"/paper/winner-takes-it-all-training-performant-rl-1","slug":"winner-takes-it-all-training-performant-rl-1","title":"Winner Takes It All: Training Performant RL Populations for Combinatorial Optimization","date":"2022-10-07","arxiv_id":"2210.03475","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/winner-takes-it-all-training-performant-rl-1#ran","syntology_url":"https://syntology.ai/paper/2210.03475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03475"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-based-evasion","slug":"deep-reinforcement-learning-based-evasion","title":"Deep Reinforcement Learning based Evasion Generative Adversarial Network for Botnet Detection","date":"2022-10-06","arxiv_id":"2210.02840","repositories_listed":1,"syntology":null},{"url":"/paper/neuroevolution-is-a-competitive-alternative","slug":"neuroevolution-is-a-competitive-alternative","title":"Neuroevolution is a Competitive Alternative to Reinforcement Learning for Skill Discovery","date":"2022-10-06","arxiv_id":"2210.03516","repositories_listed":1,"syntology":null},{"url":"/paper/rainier-reinforced-knowledge-introspector-for","slug":"rainier-reinforced-knowledge-introspector-for","title":"Rainier: Reinforced Knowledge Introspector for Commonsense Question Answering","date":"2022-10-06","arxiv_id":"2210.03078","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":6,"n_honours":4,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/rainier-reinforced-knowledge-introspector-for#ran","syntology_url":"https://syntology.ai/paper/2210.03078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03078"}},"official":{"repos":["liujch1998/rainier"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/dreamshard-generalizable-embedding-table","slug":"dreamshard-generalizable-embedding-table","title":"DreamShard: Generalizable Embedding Table Placement for Recommender Systems","date":"2022-10-05","arxiv_id":"2210.02023","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dreamshard-generalizable-embedding-table#ran","syntology_url":"https://syntology.ai/paper/2210.02023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02023"}},"official":{"repos":["daochenzha/dreamshard"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-adversarial-inverse","slug":"hierarchical-adversarial-inverse","title":"Option-Aware Adversarial Inverse Reinforcement Learning for Robotic Control","date":"2022-10-05","arxiv_id":"2210.01969","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-mechanical-ventilation-treatment","slug":"towards-safe-mechanical-ventilation-treatment","title":"Towards Safe Mechanical Ventilation Treatment Using Deep Offline Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02552","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-safe-mechanical-ventilation-treatment#ran","syntology_url":"https://syntology.ai/paper/2210.02552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02552"}},"official":{"repos":["FlemmingKondrup/DeepVent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discover-deep-identification-of-symbolic-open","slug":"discover-deep-identification-of-symbolic-open","title":"DISCOVER: Deep identification of symbolically concise open-form PDEs via enhanced reinforcement-learning","date":"2022-10-04","arxiv_id":"2210.02181","repositories_listed":1,"syntology":null},{"url":"/paper/accelerate-reinforcement-learning-with-pid","slug":"accelerate-reinforcement-learning-with-pid","title":"Accelerate Reinforcement Learning with PID Controllers in the Pendulum Simulations","date":"2022-10-03","arxiv_id":"2210.00770","repositories_listed":1,"syntology":null},{"url":"/paper/cairl-a-high-performance-reinforcement","slug":"cairl-a-high-performance-reinforcement","title":"CaiRL: A High-Performance Reinforcement Learning Environment Toolkit","date":"2022-10-03","arxiv_id":"2210.01235","repositories_listed":1,"syntology":null},{"url":"/paper/latent-state-marginalization-as-a-low-cost","slug":"latent-state-marginalization-as-a-low-cost","title":"Latent State Marginalization as a Low-cost Approach for Improving Exploration","date":"2022-10-03","arxiv_id":"2210.00999","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-state-marginalization-as-a-low-cost#ran","syntology_url":"https://syntology.ai/paper/2210.00999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00999"}},"official":{"repos":["zdhnarsil/stochastic-marginal-actor-critic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/gflownets-and-variational-inference","slug":"gflownets-and-variational-inference","title":"GFlowNets and variational inference","date":"2022-10-02","arxiv_id":"2210.00580","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/gflownets-and-variational-inference#ran","syntology_url":"https://syntology.ai/paper/2210.00580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00580"}},"official":{"repos":["gfnorg/gfn_vs_hvi"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/safe-reinforcement-learning-from-pixels-using","slug":"safe-reinforcement-learning-from-pixels-using","title":"Safe Reinforcement Learning From Pixels Using a Stochastic Latent Representation","date":"2022-10-02","arxiv_id":"2210.01801","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-from-pixels-using#ran","syntology_url":"https://syntology.ai/paper/2210.01801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.01801"}},"official":{"repos":["safe-slac/safe-slac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/b2rl-an-open-source-dataset-for-building","slug":"b2rl-an-open-source-dataset-for-building","title":"B2RL: An open-source Dataset for Building Batch Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15626","repositories_listed":1,"syntology":null},{"url":"/paper/improving-policy-learning-via-language","slug":"improving-policy-learning-via-language","title":"Improving Policy Learning via Language Dynamics Distillation","date":"2022-09-30","arxiv_id":"2210.00066","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/improving-policy-learning-via-language#ran","syntology_url":"https://syntology.ai/paper/2210.00066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00066"}},"official":{"repos":["vzhong/language-dynamics-distillation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/s2p-state-conditioned-image-synthesis-for","slug":"s2p-state-conditioned-image-synthesis-for","title":"S2P: State-conditioned Image Synthesis for Data Augmentation in Offline Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15256","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/s2p-state-conditioned-image-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2209.15256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15256"}},"official":{"repos":["dsshim0125/s2p"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-exploration-method-for-reinforcement","slug":"safe-exploration-method-for-reinforcement","title":"Safe Exploration Method for Reinforcement Learning under Existence of Disturbance","date":"2022-09-30","arxiv_id":"2209.15452","repositories_listed":1,"syntology":null},{"url":"/paper/does-zero-shot-reinforcement-learning-exist","slug":"does-zero-shot-reinforcement-learning-exist","title":"Does Zero-Shot Reinforcement Learning Exist?","date":"2022-09-29","arxiv_id":"2209.14935","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-zero-shot-reinforcement-learning-exist#ran","syntology_url":"https://syntology.ai/paper/2209.14935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14935"}},"official":{"repos":["facebookresearch/controllable_agent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-via-high","slug":"offline-reinforcement-learning-via-high","title":"Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling","date":"2022-09-29","arxiv_id":"2209.14548","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-laws-for-a-multi-agent-reinforcement","slug":"scaling-laws-for-a-multi-agent-reinforcement","title":"Scaling Laws for a Multi-Agent Reinforcement Learning Model","date":"2022-09-29","arxiv_id":"2210.00849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-a-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.00849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00849"}},"official":{"repos":["orenneumann/alphazero-scaling-laws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-but-strong-baseline-for-online","slug":"a-simple-but-strong-baseline-for-online","title":"A simple but strong baseline for online continual learning: Repeated Augmented Rehearsal","date":"2022-09-28","arxiv_id":"2209.13917","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-posterior-sampling-for-1","slug":"optimistic-posterior-sampling-for-1","title":"Optimistic Posterior Sampling for Reinforcement Learning with Few Samples and Tight Guarantees","date":"2022-09-28","arxiv_id":"2209.14414","repositories_listed":1,"syntology":null},{"url":"/paper/pareto-actor-critic-for-equilibrium-selection","slug":"pareto-actor-critic-for-equilibrium-selection","title":"Pareto Actor-Critic for Equilibrium Selection in Multi-Agent Reinforcement Learning","date":"2022-09-28","arxiv_id":"2209.14344","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-tensor-networks","slug":"reinforcement-learning-with-tensor-networks","title":"Combining Reinforcement Learning and Tensor Networks, with an Application to Dynamical Large Deviations","date":"2022-09-28","arxiv_id":"2209.14089","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-transformer-in-reinforcement","slug":"exploiting-transformer-in-reinforcement","title":"Exploiting Transformer in Sparse Reward Reinforcement Learning for Interpretable Temporal Logic Motion Planning","date":"2022-09-27","arxiv_id":"2209.13220","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-affordance-learning-for-robotic","slug":"end-to-end-affordance-learning-for-robotic","title":"End-to-End Affordance Learning for Robotic Manipulation","date":"2022-09-26","arxiv_id":"2209.12941","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-meta-reinforcement-learning-using","slug":"enhanced-meta-reinforcement-learning-using","title":"Enhanced Meta Reinforcement Learning using Demonstrations in Sparse Reward Environments","date":"2022-09-26","arxiv_id":"2209.13048","repositories_listed":1,"syntology":null},{"url":"/paper/training-efficient-controllers-via-analytic","slug":"training-efficient-controllers-via-analytic","title":"Training Efficient Controllers via Analytic Policy Gradient","date":"2022-09-26","arxiv_id":"2209.13052","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-reward-shaping-for-a-robotic","slug":"unsupervised-reward-shaping-for-a-robotic","title":"Unsupervised Reward Shaping for a Robotic Sequential Picking Task from Visual Observations in a Logistics Scenario","date":"2022-09-25","arxiv_id":"2209.12350","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-via-model","slug":"explainable-reinforcement-learning-via-model","title":"Explainable Reinforcement Learning via Model Transforms","date":"2022-09-24","arxiv_id":"2209.12006","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/explainable-reinforcement-learning-via-model#ran","syntology_url":"https://syntology.ai/paper/2209.12006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12006"}},"official":{"repos":["sarah-keren/rlpe"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/unsupervised-model-based-pre-training-for","slug":"unsupervised-model-based-pre-training-for","title":"Mastering the Unsupervised Reinforcement Learning Benchmark from Pixels","date":"2022-09-24","arxiv_id":"2209.12016","repositories_listed":1,"syntology":{"n":24,"n_ran":17,"n_constructed":14,"n_ran_checked":15,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"17 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/unsupervised-model-based-pre-training-for#ran","syntology_url":"https://syntology.ai/paper/2209.12016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12016"}},"official":{"repos":["mazpie/mastering-urlb"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/an-investigation-of-the-bias-variance","slug":"an-investigation-of-the-bias-variance","title":"An Investigation of the Bias-Variance Tradeoff in Meta-Gradients","date":"2022-09-22","arxiv_id":"2209.11303","repositories_listed":1,"syntology":null},{"url":"/paper/identifiability-and-generalizability-from","slug":"identifiability-and-generalizability-from","title":"Identifiability and generalizability from multiple experts in Inverse Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.10974","repositories_listed":1,"syntology":null},{"url":"/paper/pretraining-the-vision-transformer-using-self","slug":"pretraining-the-vision-transformer-using-self","title":"Pretraining the Vision Transformer using self-supervised methods for vision based Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.10901","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-decentralized-deep-reinforcement","slug":"hierarchical-decentralized-deep-reinforcement","title":"Hierarchical Decentralized Deep Reinforcement Learning Architecture for a Simulated Four-Legged Agent","date":"2022-09-21","arxiv_id":"2210.08003","repositories_listed":1,"syntology":null},{"url":"/paper/lcrl-certified-policy-synthesis-via-logically","slug":"lcrl-certified-policy-synthesis-via-logically","title":"LCRL: Certified Policy Synthesis via Logically-Constrained Reinforcement Learning","date":"2022-09-21","arxiv_id":"2209.10341","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-discrete-soft-actor-critic","slug":"revisiting-discrete-soft-actor-critic","title":"Revisiting Discrete Soft Actor-Critic","date":"2022-09-21","arxiv_id":"2209.10081","repositories_listed":1,"syntology":null},{"url":"/paper/a-joint-imitation-reinforcement-learning","slug":"a-joint-imitation-reinforcement-learning","title":"A Joint Imitation-Reinforcement Learning Framework for Reduced Baseline Regret","date":"2022-09-20","arxiv_id":"2209.09446","repositories_listed":1,"syntology":null},{"url":"/paper/bome-bilevel-optimization-made-easy-a-simple","slug":"bome-bilevel-optimization-made-easy-a-simple","title":"BOME! Bilevel Optimization Made Easy: A Simple First-Order Approach","date":"2022-09-19","arxiv_id":"2209.08709","repositories_listed":1,"syntology":null},{"url":"/paper/latent-plans-for-task-agnostic-offline","slug":"latent-plans-for-task-agnostic-offline","title":"Latent Plans for Task-Agnostic Offline Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.08959","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-walk-by-steering-perceptive","slug":"learning-to-walk-by-steering-perceptive","title":"Learning to Walk by Steering: Perceptive Quadrupedal Locomotion in Dynamic Environments","date":"2022-09-19","arxiv_id":"2209.09233","repositories_listed":1,"syntology":null},{"url":"/paper/man-multi-action-networks-learning","slug":"man-multi-action-networks-learning","title":"MAN: Multi-Action Networks Learning","date":"2022-09-19","arxiv_id":"2209.09329","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-interventional-robustness-in","slug":"measuring-interventional-robustness-in","title":"Measuring Interventional Robustness in Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.09058","repositories_listed":1,"syntology":null},{"url":"/paper/honor-of-kings-arena-an-environment-for","slug":"honor-of-kings-arena-an-environment-for","title":"Honor of Kings Arena: an Environment for Generalization in Competitive Reinforcement Learning","date":"2022-09-18","arxiv_id":"2209.08483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/honor-of-kings-arena-an-environment-for#ran","syntology_url":"https://syntology.ai/paper/2209.08483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08483"}},"official":{"repos":["tencent-ailab/hok_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selective-token-generation-for-few-shot-1","slug":"selective-token-generation-for-few-shot-1","title":"Selective Token Generation for Few-shot Natural Language Generation","date":"2022-09-17","arxiv_id":"2209.08206","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-natural-language-generation-for-task","slug":"adaptive-natural-language-generation-for-task","title":"Adaptive Natural Language Generation for Task-oriented Dialogue via Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.07873","repositories_listed":1,"syntology":null},{"url":"/paper/look-where-you-look-saliency-guided-q","slug":"look-where-you-look-saliency-guided-q","title":"Look where you look! Saliency-guided Q-networks for generalization in visual Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.09203","repositories_listed":1,"syntology":null}],"record_sha256":"52f994ae35e5ffd8f812a689bbaabcdf577e76bc4459e186fd72a7dd90a5385e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}