{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/ran/5","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":5,"pages_in_order":12,"rows_per_page":100,"rows":[401,500],"of":1165,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2/papers/ran/1","prev":"/task/reinforcement-learning-2/papers/ran/4","next":"/task/reinforcement-learning-2/papers/ran/6","papers":[{"url":"/paper/robust-knowledge-transfer-in-tiered-1","slug":"robust-knowledge-transfer-in-tiered-1","title":"Robust Knowledge Transfer in Tiered Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05534","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-knowledge-transfer-in-tiered-1#ran","syntology_url":"https://syntology.ai/paper/2302.05534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.05534"}},"official":{"repos":["jiaweihhuang/robust-tiered-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-recommendations-with-reinforcement","slug":"multi-task-recommendations-with-reinforcement","title":"Multi-Task Recommendations with Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03328","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-recommendations-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.03328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03328"}},"official":{"repos":["applied-machine-learning-lab/rmtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-large-language-models-in","slug":"grounding-large-language-models-in","title":"Grounding Large Language Models in Interactive Environments with Online Reinforcement Learning","date":"2023-02-06","arxiv_id":"2302.02662","repositories_listed":3,"syntology":{"n":12,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/grounding-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2302.02662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02662"}},"official":{"repos":["clementromac/lamorel","flowersteam/grounding_llms_with_online_rl","flowersteam/lamorel"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-online-reinforcement-learning-with","slug":"efficient-online-reinforcement-learning-with","title":"Efficient Online Reinforcement Learning with Offline Data","date":"2023-02-06","arxiv_id":"2302.02948","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2302.02948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02948"}},"official":{"repos":["ikostrikov/rlpd"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/locally-constrained-policy-optimization-for","slug":"locally-constrained-policy-optimization-for","title":"Online Reinforcement Learning in Non-Stationary Context-Driven Environments","date":"2023-02-04","arxiv_id":"2302.02182","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/locally-constrained-policy-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2302.02182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02182"}},"official":{"repos":["lcpo-rl/lcpo","pouyahmdn/lcpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-optimize-for-reinforcement","slug":"learning-to-optimize-for-reinforcement","title":"Learning to Optimize for Reinforcement Learning","date":"2023-02-03","arxiv_id":"2302.01470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-optimize-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.01470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01470"}},"official":{"repos":["sail-sg/optim4rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-expansion-for-bridging-offline-to","slug":"policy-expansion-for-bridging-offline-to","title":"Policy Expansion for Bridging Offline-to-Online Reinforcement Learning","date":"2023-02-02","arxiv_id":"2302.00935","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-expansion-for-bridging-offline-to#ran","syntology_url":"https://syntology.ai/paper/2302.00935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00935"}},"official":{"repos":["haichao-zhang/pex"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/internally-rewarded-reinforcement-learning","slug":"internally-rewarded-reinforcement-learning","title":"Internally Rewarded Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00270","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internally-rewarded-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00270"}},"official":{"repos":["mengdi-li/internally-rewarded-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/off-the-grid-marl-a-framework-for-dataset","slug":"off-the-grid-marl-a-framework-for-dataset","title":"Off-the-Grid MARL: Datasets with Baselines for Offline Multi-Agent Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00521","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-the-grid-marl-a-framework-for-dataset#ran","syntology_url":"https://syntology.ai/paper/2302.00521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00521"}},"official":{"repos":["instadeepai/og-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-multi-task-reinforcement-learning","slug":"efficient-multi-task-reinforcement-learning","title":"QMP: Q-switch Mixture of Policies for Multi-Task Behavior Sharing","date":"2023-02-01","arxiv_id":"2302.00671","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-multi-task-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00671"}},"official":null}},{"url":"/paper/collaborating-with-language-models-for","slug":"collaborating-with-language-models-for","title":"Collaborating with language models for embodied reasoning","date":"2023-02-01","arxiv_id":"2302.00763","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborating-with-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2302.00763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00763"}},"official":null}},{"url":"/paper/execution-based-code-generation-using-deep","slug":"execution-based-code-generation-using-deep","title":"Execution-based Code Generation using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13816","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/execution-based-code-generation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2301.13816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13816"}},"official":{"repos":["reddy-lab-code-research/PPOCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/importance-weighted-actor-critic-for-optimal-1","slug":"importance-weighted-actor-critic-for-optimal-1","title":"Importance Weighted Actor-Critic for Optimal Conservative Offline Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.12714","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/importance-weighted-actor-critic-for-optimal-1#ran","syntology_url":"https://syntology.ai/paper/2301.12714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12714"}},"official":{"repos":["zhuhl98/acrab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-based-policy-optimization-1","slug":"direct-preference-based-policy-optimization-1","title":"Direct Preference-based Policy Optimization without Reward Modeling","date":"2023-01-30","arxiv_id":"2301.12842","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-preference-based-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/2301.12842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12842"}},"official":{"repos":["snu-mllab/dppo"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-online-reinforcement-learning-with","slug":"guiding-online-reinforcement-learning-with","title":"Guiding Online Reinforcement Learning with Action-Free Offline Pretraining","date":"2023-01-30","arxiv_id":"2301.12876","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2301.12876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12876"}},"official":{"repos":["vision-cair/af-guide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/apac-authorized-probability-controlled-actor","slug":"apac-authorized-probability-controlled-actor","title":"Constrained Policy Optimization with Explicit Behavior Density for Offline Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12130","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/apac-authorized-probability-controlled-actor#ran","syntology_url":"https://syntology.ai/paper/2301.12130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12130"}},"official":{"repos":["evalarzj/cped"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/outcome-directed-reinforcement-learning-by","slug":"outcome-directed-reinforcement-learning-by","title":"Outcome-directed Reinforcement Learning by Uncertainty & Temporal Distance-Aware Curriculum Goal Generation","date":"2023-01-27","arxiv_id":"2301.11741","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":1,"n_ran_checked":2,"n_instrument":5,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/outcome-directed-reinforcement-learning-by#ran","syntology_url":"https://syntology.ai/paper/2301.11741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11741"}},"official":{"repos":["jaylee0301/outpace_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/efficient-trust-region-based-safe","slug":"efficient-trust-region-based-safe","title":"Trust Region-Based Safe Distributional Reinforcement Learning for Multiple Constraints","date":"2023-01-26","arxiv_id":"2301.10923","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-trust-region-based-safe#ran","syntology_url":"https://syntology.ai/paper/2301.10923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.10923"}},"official":{"repos":["rllab-snu/safe-distributional-actor-critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-aware-eligibility-traces-for-off","slug":"trajectory-aware-eligibility-traces-for-off","title":"Trajectory-Aware Eligibility Traces for Off-Policy Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11321","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trajectory-aware-eligibility-traces-for-off#ran","syntology_url":"https://syntology.ai/paper/2301.11321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11321"}},"official":{"repos":["brett-daley/trajectory-aware-etraces"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-moral-choices-in-social-dilemmas","slug":"modeling-moral-choices-in-social-dilemmas","title":"Modeling Moral Choices in Social Dilemmas with Multi-Agent Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08491","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-moral-choices-in-social-dilemmas#ran","syntology_url":"https://syntology.ai/paper/2301.08491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.08491"}},"official":{"repos":["liza-tennant/moral_choice_dyadic","Liza-Tennant/modeling_moral_choice_dyadic"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/heterogeneous-multi-robot-reinforcement","slug":"heterogeneous-multi-robot-reinforcement","title":"Heterogeneous Multi-Robot Reinforcement Learning","date":"2023-01-17","arxiv_id":"2301.07137","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/heterogeneous-multi-robot-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2301.07137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07137"}},"official":{"repos":["proroklab/hetgppo","proroklab/vectorizedmultiagentsimulator"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mastering-diverse-domains-through-world","slug":"mastering-diverse-domains-through-world","title":"Mastering Diverse Domains through World Models","date":"2023-01-10","arxiv_id":"2301.04104","repositories_listed":7,"syntology":{"n":34,"n_ran":21,"n_constructed":0,"n_ran_checked":16,"n_instrument":5,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 5 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/mastering-diverse-domains-through-world#ran","syntology_url":"https://syntology.ai/paper/2301.04104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.04104"}},"official":{"repos":["danijar/dreamerv3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/orbit-a-unified-simulation-framework-for","slug":"orbit-a-unified-simulation-framework-for","title":"Orbit: A Unified Simulation Framework for Interactive Robot Learning Environments","date":"2023-01-10","arxiv_id":"2301.04195","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/orbit-a-unified-simulation-framework-for#ran","syntology_url":"https://syntology.ai/paper/2301.04195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.04195"}},"official":{"repos":["NVIDIA-Omniverse/Orbit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-collective-intelligence-from-massive","slug":"emergent-collective-intelligence-from-massive","title":"Emergent collective intelligence from massive-agent cooperation and competition","date":"2023-01-04","arxiv_id":"2301.01609","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergent-collective-intelligence-from-massive#ran","syntology_url":"https://syntology.ai/paper/2301.01609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.01609"}},"official":{"repos":["hanmochen/lux-open"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/second-thoughts-are-best-learning-to-re-align","slug":"second-thoughts-are-best-learning-to-re-align","title":"Second Thoughts are Best: Learning to Re-Align With Human Values from Text Edits","date":"2023-01-01","arxiv_id":"2301.00355","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/second-thoughts-are-best-learning-to-re-align#ran","syntology_url":"https://syntology.ai/paper/2301.00355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00355"}},"official":null}},{"url":"/paper/on-pathologies-in-kl-regularized-1","slug":"on-pathologies-in-kl-regularized-1","title":"On Pathologies in KL-Regularized Reinforcement Learning from Expert Demonstrations","date":"2022-12-28","arxiv_id":"2212.13936","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-pathologies-in-kl-regularized-1#ran","syntology_url":"https://syntology.ai/paper/2212.13936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13936"}},"official":{"repos":["conglu1997/nppac"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameters-in-contextual-rl-are-highly","slug":"hyperparameters-in-contextual-rl-are-highly","title":"Hyperparameters in Contextual RL are Highly Situational","date":"2022-12-21","arxiv_id":"2212.10876","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameters-in-contextual-rl-are-highly#ran","syntology_url":"https://syntology.ai/paper/2212.10876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10876"}},"official":{"repos":["automl-private/crl_hpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-prompts-for-text-to-image-1","slug":"optimizing-prompts-for-text-to-image-1","title":"Optimizing Prompts for Text-to-Image Generation","date":"2022-12-19","arxiv_id":"2212.09611","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/optimizing-prompts-for-text-to-image-1#ran","syntology_url":"https://syntology.ai/paper/2212.09611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09611"}},"official":{"repos":["microsoft/lmops"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/effects-of-spectral-normalization-in-multi","slug":"effects-of-spectral-normalization-in-multi","title":"Effects of Spectral Normalization in Multi-agent Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effects-of-spectral-normalization-in-multi#ran","syntology_url":"https://syntology.ai/paper/2212.05331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05331"}},"official":{"repos":["kinalmehta/epymarl_spectral"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/welfare-and-fairness-in-multi-objective","slug":"welfare-and-fairness-in-multi-objective","title":"Welfare and Fairness in Multi-objective Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.01382","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/welfare-and-fairness-in-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2212.01382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01382"}},"official":{"repos":["MuhangTian/Fair-MORL-AAMAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/quantile-constrained-reinforcement-learning-a","slug":"quantile-constrained-reinforcement-learning-a","title":"Quantile Constrained Reinforcement Learning: A Reinforcement Learning Framework Constraining Outage Probability","date":"2022-11-28","arxiv_id":"2211.15034","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quantile-constrained-reinforcement-learning-a#ran","syntology_url":"https://syntology.ai/paper/2211.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15034"}},"official":{"repos":["wyjung0625/qcpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-representation-of-asymmetrical","slug":"improved-representation-of-asymmetrical","title":"Improved Representation of Asymmetrical Distances with Interval Quasimetric Embeddings","date":"2022-11-28","arxiv_id":"2211.15120","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improved-representation-of-asymmetrical#ran","syntology_url":"https://syntology.ai/paper/2211.15120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15120"}},"official":{"repos":["quasimetric-learning/torch-quasimetric"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/applying-deep-reinforcement-learning-to-the","slug":"applying-deep-reinforcement-learning-to-the","title":"Applying Deep Reinforcement Learning to the HP Model for Protein Structure Prediction","date":"2022-11-27","arxiv_id":"2211.14939","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/applying-deep-reinforcement-learning-to-the#ran","syntology_url":"https://syntology.ai/paper/2211.14939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14939"}},"official":{"repos":["compsoftmatterbiophysics-cityu-hk/applying-drl-to-hp-model-for-protein-structure-prediction"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-autoencoding-for-scalable-and","slug":"masked-autoencoding-for-scalable-and","title":"Masked Autoencoding for Scalable and Generalizable Decision Making","date":"2022-11-23","arxiv_id":"2211.12740","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-autoencoding-for-scalable-and#ran","syntology_url":"https://syntology.ai/paper/2211.12740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12740"}},"official":{"repos":["fangchenliu/maskdp_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-clean-imitation-learning","slug":"imitation-clean-imitation-learning","title":"imitation: Clean Imitation Learning Implementations","date":"2022-11-22","arxiv_id":"2211.11972","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-clean-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2211.11972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11972"}},"official":{"repos":["HumanCompatibleAI/imitation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/efficient-meta-reinforcement-learning-for","slug":"efficient-meta-reinforcement-learning-for","title":"Efficient Meta Reinforcement Learning for Preference-based Fast Adaptation","date":"2022-11-20","arxiv_id":"2211.10861","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-meta-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10861"}},"official":{"repos":["stilwell-git/adaptation-with-noisy-oracle"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/let-offline-rl-flow-training-conservative","slug":"let-offline-rl-flow-training-conservative","title":"Let Offline RL Flow: Training Conservative Agents in the Latent Space of Normalizing Flows","date":"2022-11-20","arxiv_id":"2211.11096","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/let-offline-rl-flow-training-conservative#ran","syntology_url":"https://syntology.ai/paper/2211.11096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11096"}},"official":{"repos":["tinkoff-ai/cnf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-conditioned-reinforcement-learning","slug":"language-conditioned-reinforcement-learning","title":"Language-Conditioned Reinforcement Learning to Solve Misunderstandings with Action Corrections","date":"2022-11-18","arxiv_id":"2211.10168","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-conditioned-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.10168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10168"}},"official":{"repos":["frankroeder/lanro-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-defense-against-backdoor-policies-in","slug":"provable-defense-against-backdoor-policies-in","title":"Provable Defense against Backdoor Policies in Reinforcement Learning","date":"2022-11-18","arxiv_id":"2211.10530","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/provable-defense-against-backdoor-policies-in#ran","syntology_url":"https://syntology.ai/paper/2211.10530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10530"}},"official":{"repos":["skbharti/provable-defense-in-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-data-driven-offline-simulations-for","slug":"towards-data-driven-offline-simulations-for","title":"Towards Data-Driven Offline Simulations for Online Reinforcement Learning","date":"2022-11-14","arxiv_id":"2211.07614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-data-driven-offline-simulations-for#ran","syntology_url":"https://syntology.ai/paper/2211.07614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07614"}},"official":{"repos":["microsoft/rl-offline-simulation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finrl-meta-market-environments-and-benchmarks","slug":"finrl-meta-market-environments-and-benchmarks","title":"FinRL-Meta: Market Environments and Benchmarks for Data-Driven Financial Reinforcement Learning","date":"2022-11-06","arxiv_id":"2211.03107","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finrl-meta-market-environments-and-benchmarks#ran","syntology_url":"https://syntology.ai/paper/2211.03107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03107"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-benefits-of-model-based-generalization-in","slug":"the-benefits-of-model-based-generalization-in","title":"The Benefits of Model-Based Generalization in Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02222","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-benefits-of-model-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2211.02222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02222"}},"official":{"repos":["kenjyoung/model_generalization_code_supplement"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diversity-based-deep-reinforcement-learning","slug":"diversity-based-deep-reinforcement-learning","title":"Diversity-based Deep Reinforcement Learning Towards Multidimensional Difficulty for Fighting Game AI","date":"2022-11-04","arxiv_id":"2211.02759","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diversity-based-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.02759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02759"}},"official":{"repos":["emily-halina/brisket"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-multi-agent-reinforcement-learning-2","slug":"scalable-multi-agent-reinforcement-learning-2","title":"Scalable Multi-Agent Reinforcement Learning through Intelligent Information Aggregation","date":"2022-11-03","arxiv_id":"2211.02127","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-multi-agent-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2211.02127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02127"}},"official":{"repos":["nsidn98/informarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-13","slug":"multi-agent-reinforcement-learning-for-13","title":"Multi-Agent Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-11-02","arxiv_id":"2211.00801","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-13#ran","syntology_url":"https://syntology.ai/paper/2211.00801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00801"}},"official":{"repos":["011235813/marl-amr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-prior-representation-learning-for","slug":"behavior-prior-representation-learning-for","title":"Behavior Prior Representation learning for Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.00863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/behavior-prior-representation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.00863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00863"}},"official":{"repos":["bit1029public/offline_bpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-generator-offline-reinforcement-learning","slug":"dual-generator-offline-reinforcement-learning","title":"Dual Generator Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.01471","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-generator-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.01471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.01471"}},"official":null}},{"url":"/paper/lad-language-augmented-diffusion-for","slug":"lad-language-augmented-diffusion-for","title":"Language Control Diffusion: Efficiently Scaling through Space, Time, and Tasks","date":"2022-10-27","arxiv_id":"2210.15629","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lad-language-augmented-diffusion-for#ran","syntology_url":"https://syntology.ai/paper/2210.15629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15629"}},"official":{"repos":["ezhang7423/language-control-diffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/low-rank-modular-reinforcement-learning-via","slug":"low-rank-modular-reinforcement-learning-via","title":"Low-Rank Modular Reinforcement Learning via Muscle Synergy","date":"2022-10-26","arxiv_id":"2210.15479","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/low-rank-modular-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2210.15479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15479"}},"official":{"repos":["drdh/synergy-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/erl-re-2-efficient-evolutionary-reinforcement","slug":"erl-re-2-efficient-evolutionary-reinforcement","title":"ERL-Re$^2$: Efficient Evolutionary Reinforcement Learning with Shared State Representation and Individual Policy Representation","date":"2022-10-26","arxiv_id":"2210.17375","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erl-re-2-efficient-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.17375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.17375"}},"official":{"repos":["yeshenpy/erl-re2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-behavior-cloning-regularization-for-1","slug":"adaptive-behavior-cloning-regularization-for-1","title":"Adaptive Behavior Cloning Regularization for Stable Offline-to-Online Reinforcement Learning","date":"2022-10-25","arxiv_id":"2210.13846","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-behavior-cloning-regularization-for-1#ran","syntology_url":"https://syntology.ai/paper/2210.13846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13846"}},"official":{"repos":["zhaoyi11/adaptive_bc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/in-context-reinforcement-learning-with","slug":"in-context-reinforcement-learning-with","title":"In-context Reinforcement Learning with Algorithm Distillation","date":"2022-10-25","arxiv_id":"2210.14215","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/in-context-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2210.14215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14215"}},"official":null}},{"url":"/paper/evaluating-long-term-memory-in-3d-mazes","slug":"evaluating-long-term-memory-in-3d-mazes","title":"Evaluating Long-Term Memory in 3D Mazes","date":"2022-10-24","arxiv_id":"2210.13383","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-long-term-memory-in-3d-mazes#ran","syntology_url":"https://syntology.ai/paper/2210.13383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13383"}},"official":{"repos":["jurgisp/memory-maze"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-evolution-of-linear-regions","slug":"understanding-the-evolution-of-linear-regions","title":"Understanding the Evolution of Linear Regions in Deep Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13611","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/understanding-the-evolution-of-linear-regions#ran","syntology_url":"https://syntology.ai/paper/2210.13611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13611"}},"official":{"repos":["setarehc/deep_rl_regions"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/hypernetworks-in-meta-reinforcement-learning","slug":"hypernetworks-in-meta-reinforcement-learning","title":"Hypernetworks in Meta-Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11348","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hypernetworks-in-meta-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11348"}},"official":{"repos":["jacooba/hyper"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-lifelong-reinforcement-learning","slug":"model-based-lifelong-reinforcement-learning","title":"Model-based Lifelong Reinforcement Learning with Bayesian Exploration","date":"2022-10-20","arxiv_id":"2210.11579","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-lifelong-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11579"}},"official":{"repos":["minusadd/vblrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/on-the-feasibility-of-cross-task-transfer","slug":"on-the-feasibility-of-cross-task-transfer","title":"On the Feasibility of Cross-Task Transfer with Model-Based Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10763","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/on-the-feasibility-of-cross-task-transfer#ran","syntology_url":"https://syntology.ai/paper/2210.10763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10763"}},"official":{"repos":["mlpc-ucsd/xtra"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-ask-for-help-proactive-interventions","slug":"when-to-ask-for-help-proactive-interventions","title":"When to Ask for Help: Proactive Interventions in Autonomous Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10765","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/when-to-ask-for-help-proactive-interventions#ran","syntology_url":"https://syntology.ai/paper/2210.10765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10765"}},"official":{"repos":["tajwarfahim/proactive_interventions"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-black-box-reinforcement-learning-with","slug":"deep-black-box-reinforcement-learning-with","title":"Deep Black-Box Reinforcement Learning with Movement Primitives","date":"2022-10-18","arxiv_id":"2210.09622","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-black-box-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2210.09622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09622"}},"official":{"repos":["ALRhub/fancy_gym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-value-function-learning-for","slug":"rethinking-value-function-learning-for","title":"Rethinking Value Function Learning for Generalization in Reinforcement Learning","date":"2022-10-18","arxiv_id":"2210.09960","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/rethinking-value-function-learning-for#ran","syntology_url":"https://syntology.ai/paper/2210.09960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09960"}},"official":{"repos":["snu-mllab/dcpg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/curriculum-reinforcement-learning-using","slug":"curriculum-reinforcement-learning-using","title":"Curriculum Reinforcement Learning using Optimal Transport via Gradual Domain Adaptation","date":"2022-10-18","arxiv_id":"2210.10195","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/curriculum-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2210.10195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10195"}},"official":{"repos":["peidehuang/gradient"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/teacher-forcing-recovers-reward-functions-for","slug":"teacher-forcing-recovers-reward-functions-for","title":"Teacher Forcing Recovers Reward Functions for Text Generation","date":"2022-10-17","arxiv_id":"2210.08708","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teacher-forcing-recovers-reward-functions-for#ran","syntology_url":"https://syntology.ai/paper/2210.08708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08708"}},"official":{"repos":["manga-uofa/lmreward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-policy-guided-imitation-approach-for","slug":"a-policy-guided-imitation-approach-for","title":"A Policy-Guided Imitation Approach for Offline Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08323","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-policy-guided-imitation-approach-for#ran","syntology_url":"https://syntology.ai/paper/2210.08323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08323"}},"official":{"repos":["ryanxhr/por"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-the-robustness-of-pecnet","slug":"analyzing-the-robustness-of-pecnet","title":"G-PECNet: Towards a Generalizable Pedestrian Trajectory Prediction System","date":"2022-10-15","arxiv_id":"2210.09846","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/analyzing-the-robustness-of-pecnet#ran","syntology_url":"https://syntology.ai/paper/2210.09846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09846"}},"official":{"repos":["aryan-garg/pecnet-pedestrian-trajectory-prediction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mutual-information-regularized-offline-1","slug":"mutual-information-regularized-offline-1","title":"Mutual Information Regularized Offline Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07484","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mutual-information-regularized-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07484"}},"official":{"repos":["sail-sg/misa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/touplegdd-a-fine-designed-solution-of","slug":"touplegdd-a-fine-designed-solution-of","title":"ToupleGDD: A Fine-Designed Solution of Influence Maximization by Deep Reinforcement Learning","date":"2022-10-14","arxiv_id":"2210.07500","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/touplegdd-a-fine-designed-solution-of#ran","syntology_url":"https://syntology.ai/paper/2210.07500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07500"}},"official":{"repos":["Dtrycode/ToupleGDD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-model-based-reinforcement-learning-with-2","slug":"safe-model-based-reinforcement-learning-with-2","title":"Safe Model-Based Reinforcement Learning with an Uncertainty-Aware Reachability Certificate","date":"2022-10-14","arxiv_id":"2210.07553","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-model-based-reinforcement-learning-with-2#ran","syntology_url":"https://syntology.ai/paper/2210.07553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07553"}},"official":{"repos":["ManUtdMoon/Safe_MBRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-safe-deep-reinforcement-learning","slug":"model-based-safe-deep-reinforcement-learning","title":"Model-based Safe Deep Reinforcement Learning via a Constrained Proximal Policy Optimization Algorithm","date":"2022-10-14","arxiv_id":"2210.07573","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/model-based-safe-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.07573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07573"}},"official":{"repos":["akjayant/mbppol"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-mixture-of-surprises-for-unsupervised","slug":"a-mixture-of-surprises-for-unsupervised","title":"A Mixture of Surprises for Unsupervised Reinforcement Learning","date":"2022-10-13","arxiv_id":"2210.06702","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-mixture-of-surprises-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2210.06702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06702"}},"official":{"repos":["leaplabthu/moss"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/corl-research-oriented-deep-offline-1","slug":"corl-research-oriented-deep-offline-1","title":"CORL: Research-oriented Deep Offline Reinforcement Learning Library","date":"2022-10-13","arxiv_id":"2210.07105","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/corl-research-oriented-deep-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07105"}},"official":{"repos":["corl-team/CORL","hanjuku-kaso/awesome-offline-rl","tinkoff-ai/CORL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-reinforcement-learning-with-self","slug":"visual-reinforcement-learning-with-self","title":"Visual Reinforcement Learning with Self-Supervised 3D Representations","date":"2022-10-13","arxiv_id":"2210.07241","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visual-reinforcement-learning-with-self#ran","syntology_url":"https://syntology.ai/paper/2210.07241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07241"}},"official":{"repos":["YanjieZe/rl3d"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-alternating-offline","slug":"a-unified-framework-for-alternating-offline","title":"A Unified Framework for Alternating Offline Model Training and Policy Learning","date":"2022-10-12","arxiv_id":"2210.05922","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-alternating-offline#ran","syntology_url":"https://syntology.ai/paper/2210.05922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05922"}},"official":{"repos":["shentao-yang/ampl_neurips2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-adversarial-training-without","slug":"efficient-adversarial-training-without","title":"Efficient Adversarial Training without Attacking: Worst-Case-Aware Robust Reinforcement Learning","date":"2022-10-12","arxiv_id":"2210.05927","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-adversarial-training-without#ran","syntology_url":"https://syntology.ai/paper/2210.05927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05927"}},"official":{"repos":["umd-huang-lab/wocar-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-supervised-offline-reinforcement-1","slug":"semi-supervised-offline-reinforcement-1","title":"Semi-Supervised Offline Reinforcement Learning with Action-Free Trajectories","date":"2022-10-12","arxiv_id":"2210.06518","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/semi-supervised-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2210.06518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06518"}},"official":{"repos":["facebookresearch/ssorl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/dhrl-a-graph-based-approach-for-long-horizon","slug":"dhrl-a-graph-based-approach-for-long-horizon","title":"DHRL: A Graph-Based Approach for Long-Horizon and Sparse Hierarchical Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05150","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dhrl-a-graph-based-approach-for-long-horizon#ran","syntology_url":"https://syntology.ai/paper/2210.05150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05150"}},"official":null}},{"url":"/paper/leco-learnable-episodic-count-for-task","slug":"leco-learnable-episodic-count-for-task","title":"LECO: Learnable Episodic Count for Task-Specific Intrinsic Reward","date":"2022-10-11","arxiv_id":"2210.05409","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":1,"n_ran_checked":1,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leco-learnable-episodic-count-for-task#ran","syntology_url":"https://syntology.ai/paper/2210.05409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05409"}},"official":{"repos":["kakaobrain/leco"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mastering-the-game-of-no-press-diplomacy-via","slug":"mastering-the-game-of-no-press-diplomacy-via","title":"Mastering the Game of No-Press Diplomacy via Human-Regularized Reinforcement Learning and Planning","date":"2022-10-11","arxiv_id":"2210.05492","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mastering-the-game-of-no-press-diplomacy-via#ran","syntology_url":"https://syntology.ai/paper/2210.05492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05492"}},"official":null}},{"url":"/paper/marllib-extending-rllib-for-multi-agent","slug":"marllib-extending-rllib-for-multi-agent","title":"MARLlib: A Scalable and Efficient Multi-agent Reinforcement Learning Library","date":"2022-10-11","arxiv_id":"2210.13708","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/marllib-extending-rllib-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2210.13708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13708"}},"official":{"repos":["replicable-marl/marllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-reinforcement-learning-1","slug":"benchmarking-reinforcement-learning-1","title":"Benchmarking Reinforcement Learning Techniques for Autonomous Navigation","date":"2022-10-10","arxiv_id":"2210.04839","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2210.04839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04839"}},"official":null}},{"url":"/paper/flexible-attention-based-multi-policy-fusion-1","slug":"flexible-attention-based-multi-policy-fusion-1","title":"Flexible Attention-Based Multi-Policy Fusion for Efficient Deep Reinforcement Learning","date":"2022-10-07","arxiv_id":"2210.03729","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flexible-attention-based-multi-policy-fusion-1#ran","syntology_url":"https://syntology.ai/paper/2210.03729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03729"}},"official":{"repos":["pascalson/kgrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distributionally-adaptive-meta-reinforcement","slug":"distributionally-adaptive-meta-reinforcement","title":"Distributionally Adaptive Meta Reinforcement Learning","date":"2022-10-06","arxiv_id":"2210.03104","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distributionally-adaptive-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.03104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03104"}},"official":null}},{"url":"/paper/towards-safe-mechanical-ventilation-treatment","slug":"towards-safe-mechanical-ventilation-treatment","title":"Towards Safe Mechanical Ventilation Treatment Using Deep Offline Reinforcement Learning","date":"2022-10-05","arxiv_id":"2210.02552","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-safe-mechanical-ventilation-treatment#ran","syntology_url":"https://syntology.ai/paper/2210.02552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02552"}},"official":{"repos":["FlemmingKondrup/DeepVent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stateful-active-facilitator-coordination-and","slug":"stateful-active-facilitator-coordination-and","title":"Stateful active facilitator: Coordination and Environmental Heterogeneity in Cooperative Multi-Agent Reinforcement Learning","date":"2022-10-04","arxiv_id":"2210.03022","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stateful-active-facilitator-coordination-and#ran","syntology_url":"https://syntology.ai/paper/2210.03022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03022"}},"official":{"repos":["jaggbow/saf","veds12/hecogrid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-from-pixels-using","slug":"safe-reinforcement-learning-from-pixels-using","title":"Safe Reinforcement Learning From Pixels Using a Stochastic Latent Representation","date":"2022-10-02","arxiv_id":"2210.01801","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-from-pixels-using#ran","syntology_url":"https://syntology.ai/paper/2210.01801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.01801"}},"official":{"repos":["safe-slac/safe-slac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/s2p-state-conditioned-image-synthesis-for","slug":"s2p-state-conditioned-image-synthesis-for","title":"S2P: State-conditioned Image Synthesis for Data Augmentation in Offline Reinforcement Learning","date":"2022-09-30","arxiv_id":"2209.15256","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/s2p-state-conditioned-image-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2209.15256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15256"}},"official":{"repos":["dsshim0125/s2p"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/does-zero-shot-reinforcement-learning-exist","slug":"does-zero-shot-reinforcement-learning-exist","title":"Does Zero-Shot Reinforcement Learning Exist?","date":"2022-09-29","arxiv_id":"2209.14935","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-zero-shot-reinforcement-learning-exist#ran","syntology_url":"https://syntology.ai/paper/2209.14935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14935"}},"official":{"repos":["facebookresearch/controllable_agent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-laws-for-a-multi-agent-reinforcement","slug":"scaling-laws-for-a-multi-agent-reinforcement","title":"Scaling Laws for a Multi-Agent Reinforcement Learning Model","date":"2022-09-29","arxiv_id":"2210.00849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-a-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.00849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00849"}},"official":{"repos":["orenneumann/alphazero-scaling-laws"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-model-based-pre-training-for","slug":"unsupervised-model-based-pre-training-for","title":"Mastering the Unsupervised Reinforcement Learning Benchmark from Pixels","date":"2022-09-24","arxiv_id":"2209.12016","repositories_listed":1,"syntology":{"n":24,"n_ran":17,"n_constructed":14,"n_ran_checked":15,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"17 ran (of which 14 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/unsupervised-model-based-pre-training-for#ran","syntology_url":"https://syntology.ai/paper/2209.12016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12016"}},"official":{"repos":["mazpie/mastering-urlb"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":14,"n_ran_no_instrument_failure":15,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/on-efficient-reinforcement-learning-for-full","slug":"on-efficient-reinforcement-learning-for-full","title":"On Efficient Reinforcement Learning for Full-length Game of StarCraft II","date":"2022-09-23","arxiv_id":"2209.11553","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-efficient-reinforcement-learning-for-full#ran","syntology_url":"https://syntology.ai/paper/2209.11553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.11553"}},"official":{"repos":["liuruoze/hiernet-sc2","liuruoze/mini-AlphaStar"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/honor-of-kings-arena-an-environment-for","slug":"honor-of-kings-arena-an-environment-for","title":"Honor of Kings Arena: an Environment for Generalization in Competitive Reinforcement Learning","date":"2022-09-18","arxiv_id":"2209.08483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/honor-of-kings-arena-an-environment-for#ran","syntology_url":"https://syntology.ai/paper/2209.08483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08483"}},"official":{"repos":["tencent-ailab/hok_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-1","slug":"deep-reinforcement-learning-for-1","title":"Deep Reinforcement Learning for Cryptocurrency Trading: Practical Approach to Address Backtest Overfitting","date":"2022-09-12","arxiv_id":"2209.05559","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2209.05559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.05559"}},"official":null}},{"url":"/paper/white-box-adversarial-policies-in-deep","slug":"white-box-adversarial-policies-in-deep","title":"Red Teaming with Mind Reading: White-Box Adversarial Policies Against RL Agents","date":"2022-09-05","arxiv_id":"2209.02167","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/white-box-adversarial-policies-in-deep#ran","syntology_url":"https://syntology.ai/paper/2209.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.02167"}},"official":{"repos":["thestephencasper/lm_white_box_attacks","thestephencasper/white_box_rarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-sample-efficient-world","slug":"transformers-are-sample-efficient-world","title":"Transformers are Sample-Efficient World Models","date":"2022-09-01","arxiv_id":"2209.00588","repositories_listed":2,"syntology":{"n":26,"n_ran":17,"n_constructed":13,"n_ran_checked":16,"n_instrument":1,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":15,"n_pointer_only":26,"phrase":"17 ran (of which 13 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformers-are-sample-efficient-world#ran","syntology_url":"https://syntology.ai/paper/2209.00588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00588"}},"official":{"repos":["eloialonso/iris"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":13,"n_ran_no_instrument_failure":16,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/style-agnostic-reinforcement-learning","slug":"style-agnostic-reinforcement-learning","title":"Style-Agnostic Reinforcement Learning","date":"2022-08-31","arxiv_id":"2208.14863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/style-agnostic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2208.14863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14863"}},"official":{"repos":["postech-cvlab/style-agnostic-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-representation-learning-in-deep","slug":"unsupervised-representation-learning-in-deep","title":"Unsupervised Representation Learning in Deep Reinforcement Learning: A Review","date":"2022-08-27","arxiv_id":"2208.14226","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-representation-learning-in-deep#ran","syntology_url":"https://syntology.ai/paper/2208.14226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.14226"}},"official":{"repos":["nicob15/state_representation_learning_methods"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/get-it-in-writing-formal-contracts-mitigate","slug":"get-it-in-writing-formal-contracts-mitigate","title":"Formal Contracts Mitigate Social Dilemmas in Multi-Agent RL","date":"2022-08-22","arxiv_id":"2208.10469","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/get-it-in-writing-formal-contracts-mitigate#ran","syntology_url":"https://syntology.ai/paper/2208.10469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10469"}},"official":{"repos":["algorithmic-alignment-lab/contracts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pd-morl-preference-driven-multi-objective","slug":"pd-morl-preference-driven-multi-objective","title":"PD-MORL: Preference-Driven Multi-Objective Reinforcement Learning Algorithm","date":"2022-08-16","arxiv_id":"2208.07914","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pd-morl-preference-driven-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2208.07914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.07914"}},"official":{"repos":["tbasaklar/PDMORL-Preference-Driven-Multi-Objective-Reinforcement-Learning-Algorithm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"3906c7e19c6611da0b6123795566106130a85bc8059775d7c2b691e69ea90cb1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}