{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/5","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":135,"rows_per_page":100,"rows":[401,500],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/4","next":"/task/reinforcement-learning-2/papers/6","papers":[{"url":"/paper/hierarchical-multi-agent-reinforcement","slug":"hierarchical-multi-agent-reinforcement","title":"Hierarchical Multi-Agent Reinforcement Learning for Air Combat Maneuvering","date":"2023-09-20","arxiv_id":"2309.11247","repositories_listed":2,"syntology":null},{"url":"/paper/a-bayesian-approach-to-robust-inverse","slug":"a-bayesian-approach-to-robust-inverse","title":"A Bayesian Approach to Robust Inverse Reinforcement Learning","date":"2023-09-15","arxiv_id":"2309.08571","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bayesian-approach-to-robust-inverse#ran","syntology_url":"https://syntology.ai/paper/2309.08571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.08571"}},"official":{"repos":["ran-weii/bmirl_tf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-control-of-self-assembly-of","slug":"dynamic-control-of-self-assembly-of","title":"Dynamic control of self-assembly of quasicrystalline structures through reinforcement learning","date":"2023-09-13","arxiv_id":"2309.06869","repositories_listed":2,"syntology":null},{"url":"/paper/offline-prompt-evaluation-and-optimization","slug":"offline-prompt-evaluation-and-optimization","title":"Query-Dependent Prompt Evaluation and Optimization with Offline Inverse RL","date":"2023-09-13","arxiv_id":"2309.06553","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-prompt-evaluation-and-optimization#ran","syntology_url":"https://syntology.ai/paper/2309.06553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06553"}},"official":{"repos":["holarissun/prompt-oirl","vanderschaarlab/prompt-oirl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shielded-reinforcement-learning-for-hybrid","slug":"shielded-reinforcement-learning-for-hybrid","title":"Shielded Reinforcement Learning for Hybrid Systems","date":"2023-08-28","arxiv_id":"2308.14424","repositories_listed":2,"syntology":null},{"url":"/paper/aligning-language-models-with-offline","slug":"aligning-language-models-with-offline","title":"Aligning Language Models with Offline Learning from Human Feedback","date":"2023-08-23","arxiv_id":"2308.12050","repositories_listed":2,"syntology":null},{"url":"/paper/dyadic-reinforcement-learning","slug":"dyadic-reinforcement-learning","title":"Dyadic Reinforcement Learning","date":"2023-08-15","arxiv_id":"2308.07843","repositories_listed":2,"syntology":null},{"url":"/paper/variations-on-the-reinforcement-learning","slug":"variations-on-the-reinforcement-learning","title":"Variations on the Reinforcement Learning performance of Blackjack","date":"2023-08-09","arxiv_id":"2308.07329","repositories_listed":2,"syntology":null},{"url":"/paper/esrl-efficient-sampling-based-reinforcement","slug":"esrl-efficient-sampling-based-reinforcement","title":"ESRL: Efficient Sampling-based Reinforcement Learning for Sequence Generation","date":"2023-08-04","arxiv_id":"2308.02223","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/esrl-efficient-sampling-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2308.02223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02223"}},"official":{"repos":["wangclnlp/DeepSpeed-Chat-Extension"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-offline-reinforcement-learning-1","slug":"benchmarking-offline-reinforcement-learning-1","title":"Benchmarking Offline Reinforcement Learning on Real-Robot Hardware","date":"2023-07-28","arxiv_id":"2307.15690","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/benchmarking-offline-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2307.15690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15690"}},"official":{"repos":["rr-learning/trifinger-rl-example","rr-learning/trifinger_rl_datasets"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/rlcd-reinforcement-learning-from-contrast","slug":"rlcd-reinforcement-learning-from-contrast","title":"RLCD: Reinforcement Learning from Contrastive Distillation for Language Model Alignment","date":"2023-07-24","arxiv_id":"2307.12950","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rlcd-reinforcement-learning-from-contrast#ran","syntology_url":"https://syntology.ai/paper/2307.12950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12950"}},"official":{"repos":["facebookresearch/rlcd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alleviating-matthew-effect-of-offline","slug":"alleviating-matthew-effect-of-offline","title":"Alleviating Matthew Effect of Offline Reinforcement Learning in Interactive Recommendation","date":"2023-07-10","arxiv_id":"2307.04571","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alleviating-matthew-effect-of-offline#ran","syntology_url":"https://syntology.ai/paper/2307.04571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04571"}},"official":{"repos":["chongminggao/dorl-codes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/model-bellman-inconsistency-for-model-based","slug":"model-bellman-inconsistency-for-model-based","title":"Model-Bellman Inconsistency for Model-based Offline Reinforcement Learning","date":"2023-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/srl-scaling-distributed-reinforcement","slug":"srl-scaling-distributed-reinforcement","title":"SRL: Scaling Distributed Reinforcement Learning to Over Ten Thousand Cores","date":"2023-06-29","arxiv_id":"2306.16688","repositories_listed":2,"syntology":null},{"url":"/paper/unified-off-policy-learning-to-rank-a","slug":"unified-off-policy-learning-to-rank-a","title":"Unified Off-Policy Learning to Rank: a Reinforcement Learning Perspective","date":"2023-06-13","arxiv_id":"2306.07528","repositories_listed":2,"syntology":null},{"url":"/paper/policy-regularization-with-dataset-constraint","slug":"policy-regularization-with-dataset-constraint","title":"Policy Regularization with Dataset Constraint for Offline Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06569","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/policy-regularization-with-dataset-constraint#ran","syntology_url":"https://syntology.ai/paper/2306.06569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06569"}},"official":{"repos":["lamda-rl/prdc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/active-vision-reinforcement-learning-under","slug":"active-vision-reinforcement-learning-under","title":"Active Vision Reinforcement Learning under Limited Visual Observability","date":"2023-06-01","arxiv_id":"2306.00975","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":4,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/active-vision-reinforcement-learning-under#ran","syntology_url":"https://syntology.ai/paper/2306.00975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00975"}},"official":{"repos":["elicassion/sugarl","elicassion/active-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/dpok-reinforcement-learning-for-fine-tuning","slug":"dpok-reinforcement-learning-for-fine-tuning","title":"DPOK: Reinforcement Learning for Fine-tuning Text-to-Image Diffusion Models","date":"2023-05-25","arxiv_id":"2305.16381","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dpok-reinforcement-learning-for-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2305.16381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16381"}},"official":{"repos":["google-research/google-research"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reticl-sequential-retrieval-of-in-context","slug":"reticl-sequential-retrieval-of-in-context","title":"RetICL: Sequential Retrieval of In-Context Examples with Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14502","repositories_listed":2,"syntology":null},{"url":"/paper/sharing-lifelong-reinforcement-learning","slug":"sharing-lifelong-reinforcement-learning","title":"Sharing Lifelong Reinforcement Learning Knowledge via Modulating Masks","date":"2023-05-18","arxiv_id":"2305.10997","repositories_listed":2,"syntology":null},{"url":"/paper/leveraging-factored-action-spaces-for","slug":"leveraging-factored-action-spaces-for","title":"Leveraging Factored Action Spaces for Efficient Offline Reinforcement Learning in Healthcare","date":"2023-05-02","arxiv_id":"2305.01738","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-factored-action-spaces-for#ran","syntology_url":"https://syntology.ai/paper/2305.01738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01738"}},"official":{"repos":["mld3/offlinerl_factoredactions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/diffmimic-efficient-motion-mimicking-with","slug":"diffmimic-efficient-motion-mimicking-with","title":"DiffMimic: Efficient Motion Mimicking with Differentiable Physics","date":"2023-04-06","arxiv_id":"2304.03274","repositories_listed":2,"syntology":null},{"url":"/paper/structured-state-space-models-for-in-context-1","slug":"structured-state-space-models-for-in-context-1","title":"Structured State Space Models for In-Context Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03982","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-state-space-models-for-in-context-1#ran","syntology_url":"https://syntology.ai/paper/2303.03982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03982"}},"official":{"repos":["luchris429/popjaxrl","luchris429/s5rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cflownets-continuous-control-with-generative","slug":"cflownets-continuous-control-with-generative","title":"CFlowNets: Continuous Control with Generative Flow Networks","date":"2023-03-04","arxiv_id":"2303.02430","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-control-autonomous-fleets-from","slug":"learning-to-control-autonomous-fleets-from","title":"Learning to Control Autonomous Fleets from Observation via Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14833","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-autonomous-fleets-from#ran","syntology_url":"https://syntology.ai/paper/2302.14833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14833"}},"official":{"repos":["carolinssc/offline-rl-amod","carolinssc/offline-rl-for-amod"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/ganterfactual-rl-understanding-reinforcement","slug":"ganterfactual-rl-understanding-reinforcement","title":"GANterfactual-RL: Understanding Reinforcement Learning Agents' Strategies through Visual Counterfactual Explanations","date":"2023-02-24","arxiv_id":"2302.12689","repositories_listed":2,"syntology":null},{"url":"/paper/combining-search-strategies-to-improve","slug":"combining-search-strategies-to-improve","title":"Reinforcement Learning for Combining Search Methods in the Calibration of Economic ABMs","date":"2023-02-23","arxiv_id":"2302.11835","repositories_listed":2,"syntology":null},{"url":"/paper/exploration-by-self-supervised-exploitation","slug":"exploration-by-self-supervised-exploitation","title":"Self-supervised network distillation: an effective approach to exploration in sparse reward environments","date":"2023-02-22","arxiv_id":"2302.11563","repositories_listed":2,"syntology":null},{"url":"/paper/demonstration-guided-reinforcement-learning-1","slug":"demonstration-guided-reinforcement-learning-1","title":"Demonstration-Guided Reinforcement Learning with Efficient Exploration for Task Automation of Surgical Robot","date":"2023-02-20","arxiv_id":"2302.09772","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-online-reinforcement-learning-with","slug":"efficient-online-reinforcement-learning-with","title":"Efficient Online Reinforcement Learning with Offline Data","date":"2023-02-06","arxiv_id":"2302.02948","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2302.02948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02948"}},"official":{"repos":["ikostrikov/rlpd"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/locally-constrained-policy-optimization-for","slug":"locally-constrained-policy-optimization-for","title":"Online Reinforcement Learning in Non-Stationary Context-Driven Environments","date":"2023-02-04","arxiv_id":"2302.02182","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/locally-constrained-policy-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2302.02182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02182"}},"official":{"repos":["lcpo-rl/lcpo","pouyahmdn/lcpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/off-the-grid-marl-a-framework-for-dataset","slug":"off-the-grid-marl-a-framework-for-dataset","title":"Off-the-Grid MARL: Datasets with Baselines for Offline Multi-Agent Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00521","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-the-grid-marl-a-framework-for-dataset#ran","syntology_url":"https://syntology.ai/paper/2302.00521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00521"}},"official":{"repos":["instadeepai/og-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-based-policy-optimization-1","slug":"direct-preference-based-policy-optimization-1","title":"Direct Preference-based Policy Optimization without Reward Modeling","date":"2023-01-30","arxiv_id":"2301.12842","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-preference-based-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/2301.12842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12842"}},"official":{"repos":["snu-mllab/dppo"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-moral-choices-in-social-dilemmas","slug":"modeling-moral-choices-in-social-dilemmas","title":"Modeling Moral Choices in Social Dilemmas with Multi-Agent Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08491","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-moral-choices-in-social-dilemmas#ran","syntology_url":"https://syntology.ai/paper/2301.08491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.08491"}},"official":{"repos":["liza-tennant/moral_choice_dyadic","Liza-Tennant/modeling_moral_choice_dyadic"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/heterogeneous-multi-robot-reinforcement","slug":"heterogeneous-multi-robot-reinforcement","title":"Heterogeneous Multi-Robot Reinforcement Learning","date":"2023-01-17","arxiv_id":"2301.07137","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/heterogeneous-multi-robot-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2301.07137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07137"}},"official":{"repos":["proroklab/hetgppo","proroklab/vectorizedmultiagentsimulator"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/asynchronous-multi-agent-reinforcement","slug":"asynchronous-multi-agent-reinforcement","title":"Asynchronous Multi-Agent Reinforcement Learning for Efficient Real-Time Multi-Robot Cooperative Exploration","date":"2023-01-09","arxiv_id":"2301.03398","repositories_listed":2,"syntology":null},{"url":"/paper/targeted-adversarial-attacks-on-deep","slug":"targeted-adversarial-attacks-on-deep","title":"Targeted Adversarial Attacks on Deep Reinforcement Learning Policies via Model Checking","date":"2022-12-10","arxiv_id":"2212.05337","repositories_listed":2,"syntology":null},{"url":"/paper/compiler-optimization-for-quantum-computing","slug":"compiler-optimization-for-quantum-computing","title":"Compiler Optimization for Quantum Computing Using Reinforcement Learning","date":"2022-12-08","arxiv_id":"2212.04508","repositories_listed":2,"syntology":null},{"url":"/paper/mo-gym-a-library-of-multi-objective","slug":"mo-gym-a-library-of-multi-objective","title":"MO-Gym: A Library of Multi-Objective Reinforcement Learning Environments","date":"2022-11-30","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/the-surprising-effectiveness-of-latent-world","slug":"the-surprising-effectiveness-of-latent-world","title":"The Effectiveness of World Models for Continual Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.15944","repositories_listed":2,"syntology":null},{"url":"/paper/pac-man-pete-an-extensible-framework-for","slug":"pac-man-pete-an-extensible-framework-for","title":"Pac-Man Pete: An extensible framework for building AI in VEX Robotics","date":"2022-11-25","arxiv_id":"2211.14385","repositories_listed":2,"syntology":null},{"url":"/paper/imitation-clean-imitation-learning","slug":"imitation-clean-imitation-learning","title":"imitation: Clean Imitation Learning Implementations","date":"2022-11-22","arxiv_id":"2211.11972","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-clean-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2211.11972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11972"}},"official":{"repos":["HumanCompatibleAI/imitation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/let-offline-rl-flow-training-conservative","slug":"let-offline-rl-flow-training-conservative","title":"Let Offline RL Flow: Training Conservative Agents in the Latent Space of Normalizing Flows","date":"2022-11-20","arxiv_id":"2211.11096","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/let-offline-rl-flow-training-conservative#ran","syntology_url":"https://syntology.ai/paper/2211.11096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11096"}},"official":{"repos":["tinkoff-ai/cnf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/libsignal-an-open-library-for-traffic-signal","slug":"libsignal-an-open-library-for-traffic-signal","title":"LibSignal: An Open Library for Traffic Signal Control","date":"2022-11-19","arxiv_id":"2211.10649","repositories_listed":2,"syntology":null},{"url":"/paper/protox-explaining-a-reinforcement-learning","slug":"protox-explaining-a-reinforcement-learning","title":"ProtoX: Explaining a Reinforcement Learning Agent via Prototyping","date":"2022-11-06","arxiv_id":"2211.03162","repositories_listed":2,"syntology":null},{"url":"/paper/scalable-multi-agent-reinforcement-learning-2","slug":"scalable-multi-agent-reinforcement-learning-2","title":"Scalable Multi-Agent Reinforcement Learning through Intelligent Information Aggregation","date":"2022-11-03","arxiv_id":"2211.02127","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-multi-agent-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2211.02127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02127"}},"official":{"repos":["nsidn98/informarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-counterfactuals-for-improving-the","slug":"causal-counterfactuals-for-improving-the","title":"Causal Counterfactuals for Improving the Robustness of Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.05551","repositories_listed":2,"syntology":null},{"url":"/paper/adaptive-behavior-cloning-regularization-for-1","slug":"adaptive-behavior-cloning-regularization-for-1","title":"Adaptive Behavior Cloning Regularization for Stable Offline-to-Online Reinforcement Learning","date":"2022-10-25","arxiv_id":"2210.13846","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-behavior-cloning-regularization-for-1#ran","syntology_url":"https://syntology.ai/paper/2210.13846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13846"}},"official":{"repos":["zhaoyi11/adaptive_bc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-lifelong-reinforcement-learning","slug":"model-based-lifelong-reinforcement-learning","title":"Model-based Lifelong Reinforcement Learning with Bayesian Exploration","date":"2022-10-20","arxiv_id":"2210.11579","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-lifelong-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11579"}},"official":{"repos":["minusadd/vblrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-multilevel-reinforcement-learning-framework","slug":"a-multilevel-reinforcement-learning-framework","title":"A Multilevel Reinforcement Learning Framework for PDE-based Control","date":"2022-10-15","arxiv_id":"2210.08400","repositories_listed":2,"syntology":null},{"url":"/paper/discovering-faster-matrix-multiplication","slug":"discovering-faster-matrix-multiplication","title":"Discovering faster matrix multiplication algorithms with reinforcement learning","date":"2022-10-05","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/stateful-active-facilitator-coordination-and","slug":"stateful-active-facilitator-coordination-and","title":"Stateful active facilitator: Coordination and Environmental Heterogeneity in Cooperative Multi-Agent Reinforcement Learning","date":"2022-10-04","arxiv_id":"2210.03022","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stateful-active-facilitator-coordination-and#ran","syntology_url":"https://syntology.ai/paper/2210.03022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03022"}},"official":{"repos":["jaggbow/saf","veds12/hecogrid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-intrinsically-motivated-exploration-in","slug":"deep-intrinsically-motivated-exploration-in","title":"Deep Intrinsically Motivated Exploration in Continuous Control","date":"2022-10-01","arxiv_id":"2210.00293","repositories_listed":2,"syntology":null},{"url":"/paper/on-efficient-reinforcement-learning-for-full","slug":"on-efficient-reinforcement-learning-for-full","title":"On Efficient Reinforcement Learning for Full-length Game of StarCraft II","date":"2022-09-23","arxiv_id":"2209.11553","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-efficient-reinforcement-learning-for-full#ran","syntology_url":"https://syntology.ai/paper/2209.11553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.11553"}},"official":{"repos":["liuruoze/hiernet-sc2","liuruoze/mini-AlphaStar"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cool-mc-a-comprehensive-tool-for","slug":"cool-mc-a-comprehensive-tool-for","title":"COOL-MC: A Comprehensive Tool for Reinforcement Learning and Model Checking","date":"2022-09-15","arxiv_id":"2209.07133","repositories_listed":2,"syntology":null},{"url":"/paper/white-box-adversarial-policies-in-deep","slug":"white-box-adversarial-policies-in-deep","title":"Red Teaming with Mind Reading: White-Box Adversarial Policies Against RL Agents","date":"2022-09-05","arxiv_id":"2209.02167","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/white-box-adversarial-policies-in-deep#ran","syntology_url":"https://syntology.ai/paper/2209.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.02167"}},"official":{"repos":["thestephencasper/lm_white_box_attacks","thestephencasper/white_box_rarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-sample-efficient-world","slug":"transformers-are-sample-efficient-world","title":"Transformers are Sample-Efficient World Models","date":"2022-09-01","arxiv_id":"2209.00588","repositories_listed":2,"syntology":{"n":26,"n_ran":17,"n_constructed":13,"n_ran_checked":16,"n_instrument":1,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":15,"n_pointer_only":26,"phrase":"17 ran (of which 13 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformers-are-sample-efficient-world#ran","syntology_url":"https://syntology.ai/paper/2209.00588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00588"}},"official":{"repos":["eloialonso/iris"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":13,"n_ran_no_instrument_failure":16,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-automated-imbalanced-learning-with","slug":"towards-automated-imbalanced-learning-with","title":"Towards Automated Imbalanced Learning with Deep Hierarchical Reinforcement Learning","date":"2022-08-26","arxiv_id":"2208.12433","repositories_listed":2,"syntology":null},{"url":"/paper/solving-royal-game-of-ur-using-reinforcement","slug":"solving-royal-game-of-ur-using-reinforcement","title":"Solving Royal Game of Ur Using Reinforcement Learning","date":"2022-08-23","arxiv_id":"2208.10669","repositories_listed":2,"syntology":null},{"url":"/paper/incorporating-rivalry-in-reinforcement-1","slug":"incorporating-rivalry-in-reinforcement-1","title":"Incorporating Rivalry in Reinforcement Learning for a Competitive Game","date":"2022-08-22","arxiv_id":"2208.10327","repositories_listed":2,"syntology":null},{"url":"/paper/metric-residual-networks-for-sample-efficient","slug":"metric-residual-networks-for-sample-efficient","title":"Metric Residual Networks for Sample Efficient Goal-Conditioned Reinforcement Learning","date":"2022-08-17","arxiv_id":"2208.08133","repositories_listed":2,"syntology":null},{"url":"/paper/bsac-bayesian-strategy-network-based-soft","slug":"bsac-bayesian-strategy-network-based-soft","title":"Bayesian Soft Actor-Critic: A Directed Acyclic Strategy Graph Based Deep Reinforcement Learning","date":"2022-08-11","arxiv_id":"2208.06033","repositories_listed":2,"syntology":null},{"url":"/paper/automating-dbscan-via-deep-reinforcement","slug":"automating-dbscan-via-deep-reinforcement","title":"Automating DBSCAN via Deep Reinforcement Learning","date":"2022-08-09","arxiv_id":"2208.04537","repositories_listed":2,"syntology":null},{"url":"/paper/hierarchical-kickstarting-for-skill-transfer","slug":"hierarchical-kickstarting-for-skill-transfer","title":"Hierarchical Kickstarting for Skill Transfer in Reinforcement Learning","date":"2022-07-23","arxiv_id":"2207.11584","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-kickstarting-for-skill-transfer#ran","syntology_url":"https://syntology.ai/paper/2207.11584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.11584"}},"official":{"repos":["ucl-dark/skillhack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/log-barriers-for-safe-black-box-optimization","slug":"log-barriers-for-safe-black-box-optimization","title":"Log Barriers for Safe Black-box Optimization with Application to Safe Reinforcement Learning","date":"2022-07-21","arxiv_id":"2207.10415","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/log-barriers-for-safe-black-box-optimization#ran","syntology_url":"https://syntology.ai/paper/2207.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10415"}},"official":{"repos":["ilnura/lb_sgd","lasgroup/lbsgd-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dgpo-discovering-multiple-strategies-with","slug":"dgpo-discovering-multiple-strategies-with","title":"DGPO: Discovering Multiple Strategies with Diversity-Guided Policy Optimization","date":"2022-07-12","arxiv_id":"2207.05631","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dgpo-discovering-multiple-strategies-with#ran","syntology_url":"https://syntology.ai/paper/2207.05631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05631"}},"official":{"repos":["OpenRL-Lab/DGPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/coderl-mastering-code-generation-through","slug":"coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.01780","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderl-mastering-code-generation-through#ran","syntology_url":"https://syntology.ai/paper/2207.01780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01780"}},"official":{"repos":["salesforce/coderl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-policy-improvement-algorithms","slug":"generalized-policy-improvement-algorithms","title":"Generalized Policy Improvement Algorithms with Theoretically Supported Sample Reuse","date":"2022-06-28","arxiv_id":"2206.13714","repositories_listed":2,"syntology":null},{"url":"/paper/video-pretraining-vpt-learning-to-act-by","slug":"video-pretraining-vpt-learning-to-act-by","title":"Video PreTraining (VPT): Learning to Act by Watching Unlabeled Online Videos","date":"2022-06-23","arxiv_id":"2206.11795","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-pretraining-vpt-learning-to-act-by#ran","syntology_url":"https://syntology.ai/paper/2206.11795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11795"}},"official":{"repos":["openai/Video-Pre-Training"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-constraint-inference-in-inverse","slug":"benchmarking-constraint-inference-in-inverse","title":"Benchmarking Constraint Inference in Inverse Reinforcement Learning","date":"2022-06-20","arxiv_id":"2206.09670","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-constraint-inference-in-inverse#ran","syntology_url":"https://syntology.ai/paper/2206.09670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09670"}},"official":{"repos":["guiliang/cirl-benchmarks-public","guiliang/icrl-benchmarks-public"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anchor-changing-regularized-natural-policy","slug":"anchor-changing-regularized-natural-policy","title":"Anchor-Changing Regularized Natural Policy Gradient for Multi-Objective Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05357","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anchor-changing-regularized-natural-policy#ran","syntology_url":"https://syntology.ai/paper/2206.05357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05357"}},"official":{"repos":["tliu1997/arnpg-morl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-self-supervised-learning-really-improve","slug":"does-self-supervised-learning-really-improve","title":"Does Self-supervised Learning Really Improve Reinforcement Learning from Pixels?","date":"2022-06-10","arxiv_id":"2206.05266","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-self-supervised-learning-really-improve#ran","syntology_url":"https://syntology.ai/paper/2206.05266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05266"}},"official":{"repos":["LostXine/elo-sac","lostxine/elo-rainbow"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/challenges-and-opportunities-in-offline","slug":"challenges-and-opportunities-in-offline","title":"Challenges and Opportunities in Offline Reinforcement Learning from Visual Observations","date":"2022-06-09","arxiv_id":"2206.04779","repositories_listed":2,"syntology":null},{"url":"/paper/task-agnostic-continual-reinforcement","slug":"task-agnostic-continual-reinforcement","title":"Task-Agnostic Continual Reinforcement Learning: Gaining Insights and Overcoming Challenges","date":"2022-05-28","arxiv_id":"2205.14495","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/task-agnostic-continual-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14495"}},"official":{"repos":["amazon-science/replay-based-recurrent-rl","amazon-research/replay-based-recurrent-rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/history-compression-via-language-models-in","slug":"history-compression-via-language-models-in","title":"History Compression via Language Models in Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12258","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/history-compression-via-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2205.12258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12258"}},"official":{"repos":["ml-jku/helm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-uncertainty-for-exploration-in-1","slug":"reward-uncertainty-for-exploration-in-1","title":"Reward Uncertainty for Exploration in Preference-based Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12401","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-uncertainty-for-exploration-in-1#ran","syntology_url":"https://syntology.ai/paper/2205.12401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12401"}},"official":null}},{"url":"/paper/distance-sensitive-offline-reinforcement","slug":"distance-sensitive-offline-reinforcement","title":"When Data Geometry Meets Deep Function: Generalizing Offline Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.11027","repositories_listed":2,"syntology":null},{"url":"/paper/reachability-constrained-reinforcement","slug":"reachability-constrained-reinforcement","title":"Reachability Constrained Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07536","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reachability-constrained-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07536"}},"official":{"repos":["mahaitongdae/Reachability_Constrained_RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/efficient-risk-averse-reinforcement-learning","slug":"efficient-risk-averse-reinforcement-learning","title":"Efficient Risk-Averse Reinforcement Learning","date":"2022-05-10","arxiv_id":"2205.05138","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-risk-averse-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2205.05138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05138"}},"official":{"repos":["ido90/CeSoR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/rambo-rl-robust-adversarial-model-based","slug":"rambo-rl-robust-adversarial-model-based","title":"RAMBO-RL: Robust Adversarial Model-Based Offline Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12581","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rambo-rl-robust-adversarial-model-based#ran","syntology_url":"https://syntology.ai/paper/2204.12581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.12581"}},"official":{"repos":["marc-rigter/rambo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue","slug":"chai-a-chatbot-ai-for-task-oriented-dialogue","title":"CHAI: A CHatbot AI for Task-Oriented Dialogue with Offline Reinforcement Learning","date":"2022-04-18","arxiv_id":"2204.08426","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue#ran","syntology_url":"https://syntology.ai/paper/2204.08426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08426"}},"official":{"repos":["siddharthverma314/chai-naacl-2022"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-on-graph-a-survey","slug":"reinforcement-learning-on-graph-a-survey","title":"Reinforcement learning on graphs: A survey","date":"2022-04-13","arxiv_id":"2204.06127","repositories_listed":2,"syntology":null},{"url":"/paper/grounding-hindsight-instructions-in-multi","slug":"grounding-hindsight-instructions-in-multi","title":"Grounding Hindsight Instructions in Multi-Goal Reinforcement Learning for Robotics","date":"2022-04-08","arxiv_id":"2204.04308","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-hindsight-instructions-in-multi#ran","syntology_url":"https://syntology.ai/paper/2204.04308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04308"}},"official":{"repos":["knowledgetechnologyuhh/hipss"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hysteresis-based-rl-robustifying","slug":"hysteresis-based-rl-robustifying","title":"Hysteresis-Based RL: Robustifying Reinforcement Learning-based Control Policies via Hybrid Control","date":"2022-04-01","arxiv_id":"2204.00654","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-action-free-pre","slug":"reinforcement-learning-with-action-free-pre","title":"Reinforcement Learning with Action-Free Pre-Training from Videos","date":"2022-03-25","arxiv_id":"2203.13880","repositories_listed":2,"syntology":null},{"url":"/paper/reasoning-about-counterfactuals-to-improve","slug":"reasoning-about-counterfactuals-to-improve","title":"Reasoning about Counterfactuals to Improve Human Inverse Reinforcement Learning","date":"2022-03-03","arxiv_id":"2203.01855","repositories_listed":2,"syntology":null},{"url":"/paper/socially-aware-robot-crowd-navigation-with","slug":"socially-aware-robot-crowd-navigation-with","title":"Intention Aware Robot Crowd Navigation with Attention-Based Interaction Graph","date":"2022-03-03","arxiv_id":"2203.01821","repositories_listed":2,"syntology":null},{"url":"/paper/combining-modular-skills-in-multitask","slug":"combining-modular-skills-in-multitask","title":"Combining Modular Skills in Multitask Learning","date":"2022-02-28","arxiv_id":"2202.13914","repositories_listed":2,"syntology":null},{"url":"/paper/quantum-deep-reinforcement-learning-for-robot","slug":"quantum-deep-reinforcement-learning-for-robot","title":"Quantum Deep Reinforcement Learning for Robot Navigation Tasks","date":"2022-02-24","arxiv_id":"2202.12180","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/quantum-deep-reinforcement-learning-for-robot#ran","syntology_url":"https://syntology.ai/paper/2202.12180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12180"}},"official":{"repos":["dfki-ric-quantum/qdrl-turtlebot-env","dfki-ric-quantum/qdrl-turtlebot-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-variational-policy-optimization","slug":"constrained-variational-policy-optimization","title":"Constrained Variational Policy Optimization for Safe Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.11927","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/constrained-variational-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2201.11927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11927"}},"official":{"repos":["liuzuxin/cvpo-safe-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/simsr-simple-distance-based-state","slug":"simsr-simple-distance-based-state","title":"SimSR: Simple Distance-based State Representation for Deep Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15303","repositories_listed":2,"syntology":null},{"url":"/paper/lane-change-decision-making-through-deep-1","slug":"lane-change-decision-making-through-deep-1","title":"Lane Change Decision-Making through Deep Reinforcement Learning","date":"2021-12-24","arxiv_id":"2112.14705","repositories_listed":2,"syntology":null},{"url":"/paper/on-the-unreasonable-efficiency-of-state-space","slug":"on-the-unreasonable-efficiency-of-state-space","title":"On the Unreasonable Efficiency of State Space Clustering in Personalization Tasks","date":"2021-12-24","arxiv_id":"2112.13141","repositories_listed":2,"syntology":null},{"url":"/paper/autonomous-reinforcement-learning-formalism-1","slug":"autonomous-reinforcement-learning-formalism-1","title":"Autonomous Reinforcement Learning: Formalism and Benchmarking","date":"2021-12-17","arxiv_id":"2112.09605","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autonomous-reinforcement-learning-formalism-1#ran","syntology_url":"https://syntology.ai/paper/2112.09605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.09605"}},"official":{"repos":["architsharma97/earl_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-share-in-multi-agent-1","slug":"learning-to-share-in-multi-agent-1","title":"Learning to Share in Multi-Agent Reinforcement Learning","date":"2021-12-16","arxiv_id":"2112.08702","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-reinforcement-learning-in","slug":"unsupervised-reinforcement-learning-in","title":"Unsupervised Reinforcement Learning in Multiple Environments","date":"2021-12-16","arxiv_id":"2112.08746","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2112.08746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.08746"}},"official":{"repos":["muttimirco/alphamepol"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-experimental-design-perspective-on-model","slug":"an-experimental-design-perspective-on-model","title":"An Experimental Design Perspective on Model-Based Reinforcement Learning","date":"2021-12-09","arxiv_id":"2112.05244","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-experimental-design-perspective-on-model#ran","syntology_url":"https://syntology.ai/paper/2112.05244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05244"}},"official":null}},{"url":"/paper/vmagent-scheduling-simulator-for","slug":"vmagent-scheduling-simulator-for","title":"VMAgent: Scheduling Simulator for Reinforcement Learning","date":"2021-12-09","arxiv_id":"2112.04785","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-symptom-inquiring-and-diagnosis-via","slug":"efficient-symptom-inquiring-and-diagnosis-via","title":"Efficient Symptom Inquiring and Diagnosis via Adaptive Alignment of Reinforcement Learning and Classification","date":"2021-12-01","arxiv_id":"2112.00733","repositories_listed":2,"syntology":null}],"record_sha256":"7d51a7fad1e0f4076bf87720bb16d7d1c60030c2bc2ffef0eab5d70459b05529","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}