{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/continuous-control/papers/ran/1","list_of":"/task/continuous-control","task":"Continuous Control","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":193,"counts":{"archive_papers_tagged":1161,"with_a_code_link":494,"where_syntology_ran_a_sample":193,"not_listed_spam_title":0,"listed":1161,"listed_where_code_ran":193,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/continuous-control/papers/ran/1","prev":null,"next":"/task/continuous-control/papers/ran/2","papers":[{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-policy-optimization-for-offline","slug":"adversarial-policy-optimization-for-offline","title":"Adversarial Policy Optimization for Offline Preference-based Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-policy-optimization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2503.05306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05306"}},"official":{"repos":["oh-lab/APPO"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/discrete-codebook-world-models-for-continuous","slug":"discrete-codebook-world-models-for-continuous","title":"Discrete Codebook World Models for Continuous Control","date":"2025-03-01","arxiv_id":"2503.00653","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discrete-codebook-world-models-for-continuous#ran","syntology_url":"https://syntology.ai/paper/2503.00653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00653"}},"official":{"repos":["aidanscannell/dcmpc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperspherical-normalization-for-scalable","slug":"hyperspherical-normalization-for-scalable","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15280","repositories_listed":0,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyperspherical-normalization-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.15280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15280"}},"official":null}},{"url":"/paper/langevin-soft-actor-critic-efficient","slug":"langevin-soft-actor-critic-efficient","title":"Langevin Soft Actor-Critic: Efficient Exploration through Uncertainty-Driven Critic Learning","date":"2025-01-29","arxiv_id":"2501.17827","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/langevin-soft-actor-critic-efficient#ran","syntology_url":"https://syntology.ai/paper/2501.17827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17827"}},"official":{"repos":["hmishfaq/lsac"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-controller-few-shot-imitation-of-unseen","slug":"meta-controller-few-shot-imitation-of-unseen","title":"Meta-Controller: Few-Shot Imitation of Unseen Embodiments and Tasks in Continuous Control","date":"2024-12-10","arxiv_id":"2412.12147","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meta-controller-few-shot-imitation-of-unseen#ran","syntology_url":"https://syntology.ai/paper/2412.12147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12147"}},"official":{"repos":["seongwoongcho/meta-controller"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cale-continuous-arcade-learning-environment","slug":"cale-continuous-arcade-learning-environment","title":"CALE: Continuous Arcade Learning Environment","date":"2024-10-31","arxiv_id":"2410.23810","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cale-continuous-arcade-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2410.23810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23810"}},"official":{"repos":["farama-foundation/arcade-learning-environment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusing-states-and-matching-scores-a-new","slug":"diffusing-states-and-matching-scores-a-new","title":"Diffusing States and Matching Scores: A New Framework for Imitation Learning","date":"2024-10-17","arxiv_id":"2410.13855","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffusing-states-and-matching-scores-a-new#ran","syntology_url":"https://syntology.ai/paper/2410.13855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13855"}},"official":{"repos":["ziqian2000/smiling"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-slow-decision-frequencies-in","slug":"overcoming-slow-decision-frequencies-in","title":"Overcoming Slow Decision Frequencies in Continuous Control: Model-Based Sequence Reinforcement Learning for Model-Free Control","date":"2024-10-11","arxiv_id":"2410.08979","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":3,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/overcoming-slow-decision-frequencies-in#ran","syntology_url":"https://syntology.ai/paper/2410.08979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08979"}},"official":{"repos":["dee0512/Temporally-Layered-Architecture"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/c-morl-multi-objective-reinforcement-learning","slug":"c-morl-multi-objective-reinforcement-learning","title":"C-MORL: Multi-Objective Reinforcement Learning through Efficient Discovery of Pareto Front","date":"2024-10-03","arxiv_id":"2410.02236","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/c-morl-multi-objective-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2410.02236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02236"}},"official":{"repos":["ruohliuq/c-morl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dmc-vb-a-benchmark-for-representation","slug":"dmc-vb-a-benchmark-for-representation","title":"DMC-VB: A Benchmark for Representation Learning for Control with Visual Distractors","date":"2024-09-26","arxiv_id":"2409.18330","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dmc-vb-a-benchmark-for-representation#ran","syntology_url":"https://syntology.ai/paper/2409.18330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18330"}},"official":{"repos":["google-deepmind/dmc_vision_benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-policy-policy-optimization","slug":"diffusion-policy-policy-optimization","title":"Diffusion Policy Policy Optimization","date":"2024-09-01","arxiv_id":"2409.00588","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-policy-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2409.00588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00588"}},"official":null}},{"url":"/paper/model-based-transfer-learning-for-contextual","slug":"model-based-transfer-learning-for-contextual","title":"Model-Based Transfer Learning for Contextual Reinforcement Learning","date":"2024-08-08","arxiv_id":"2408.04498","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-transfer-learning-for-contextual#ran","syntology_url":"https://syntology.ai/paper/2408.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04498"}},"official":{"repos":["jhoon-cho/mbtl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/last-iterate-global-convergence-of-policy","slug":"last-iterate-global-convergence-of-policy","title":"Last-Iterate Global Convergence of Policy Gradients for Constrained Reinforcement Learning","date":"2024-07-15","arxiv_id":"2407.10775","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/last-iterate-global-convergence-of-policy#ran","syntology_url":"https://syntology.ai/paper/2407.10775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10775"}},"official":null}},{"url":"/paper/aligning-diffusion-behaviors-with-q-functions","slug":"aligning-diffusion-behaviors-with-q-functions","title":"Aligning Diffusion Behaviors with Q-functions for Efficient Continuous Control","date":"2024-07-12","arxiv_id":"2407.09024","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/aligning-diffusion-behaviors-with-q-functions#ran","syntology_url":"https://syntology.ai/paper/2407.09024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09024"}},"official":{"repos":["thu-ml/efficient-diffusion-alignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/behaviour-distillation","slug":"behaviour-distillation","title":"Behaviour Distillation","date":"2024-06-21","arxiv_id":"2406.15042","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behaviour-distillation#ran","syntology_url":"https://syntology.ai/paper/2406.15042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15042"}},"official":{"repos":["flairox/behaviour-distillation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evil-evolution-strategies-for-generalisable","slug":"evil-evolution-strategies-for-generalisable","title":"EvIL: Evolution Strategies for Generalisable Imitation Learning","date":"2024-06-15","arxiv_id":"2406.11905","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evil-evolution-strategies-for-generalisable#ran","syntology_url":"https://syntology.ai/paper/2406.11905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11905"}},"official":{"repos":["SilviaSapora/evil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rrls-robust-reinforcement-learning-suite","slug":"rrls-robust-reinforcement-learning-suite","title":"RRLS : Robust Reinforcement Learning Suite","date":"2024-06-12","arxiv_id":"2406.08406","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rrls-robust-reinforcement-learning-suite#ran","syntology_url":"https://syntology.ai/paper/2406.08406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08406"}},"official":{"repos":["sureli/rrls"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/amortizing-intractable-inference-in-diffusion","slug":"amortizing-intractable-inference-in-diffusion","title":"Amortizing intractable inference in diffusion models for vision, language, and control","date":"2024-05-31","arxiv_id":"2405.20971","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/amortizing-intractable-inference-in-diffusion#ran","syntology_url":"https://syntology.ai/paper/2405.20971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20971"}},"official":{"repos":["gfnorg/diffusion-finetuning"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bigger-regularized-optimistic-scaling-for","slug":"bigger-regularized-optimistic-scaling-for","title":"Bigger, Regularized, Optimistic: scaling for compute and sample-efficient continuous control","date":"2024-05-25","arxiv_id":"2405.16158","repositories_listed":1,"syntology":{"n":20,"n_ran":11,"n_constructed":7,"n_ran_checked":10,"n_instrument":1,"n_unverified":9,"n_honours":2,"n_violates":1,"n_no_contract":7,"n_pointer_only":3,"phrase":"11 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/bigger-regularized-optimistic-scaling-for#ran","syntology_url":"https://syntology.ai/paper/2405.16158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16158"}},"official":{"repos":["naumix/BiggerRegularizedOptimistic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/how-to-leverage-diverse-demonstrations-in","slug":"how-to-leverage-diverse-demonstrations-in","title":"How to Leverage Diverse Demonstrations in Offline Imitation Learning","date":"2024-05-24","arxiv_id":"2405.17476","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/how-to-leverage-diverse-demonstrations-in#ran","syntology_url":"https://syntology.ai/paper/2405.17476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17476"}},"official":{"repos":["hansenhua/ilid-offline-imitation-learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/ollie-imitation-learning-from-offline","slug":"ollie-imitation-learning-from-offline","title":"OLLIE: Imitation Learning from Offline Pretraining to Online Finetuning","date":"2024-05-24","arxiv_id":"2405.17477","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":4,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/ollie-imitation-learning-from-offline#ran","syntology_url":"https://syntology.ai/paper/2405.17477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17477"}},"official":{"repos":["hansenhua/ollie-offline-to-online-imitation-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/offline-reinforcement-learning-from-datasets","slug":"offline-reinforcement-learning-from-datasets","title":"Offline Reinforcement Learning from Datasets with Structured Non-Stationarity","date":"2024-05-23","arxiv_id":"2405.14114","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-reinforcement-learning-from-datasets#ran","syntology_url":"https://syntology.ai/paper/2405.14114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14114"}},"official":{"repos":["johannesack/offlinerlstructurednonstationarity"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-curse-of-diversity-in-ensemble-based","slug":"the-curse-of-diversity-in-ensemble-based","title":"The Curse of Diversity in Ensemble-Based Exploration","date":"2024-05-07","arxiv_id":"2405.04342","repositories_listed":2,"syntology":{"n":16,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":13,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/the-curse-of-diversity-in-ensemble-based#ran","syntology_url":"https://syntology.ai/paper/2405.04342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04342"}},"official":{"repos":["zhixuan-lin/ensemble-rl-continuous","zhixuan-lin/ensemble-rl-discrete"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","repositories_listed":3,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rebel-reinforcement-learning-via-regressing#ran","syntology_url":"https://syntology.ai/paper/2404.16767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16767"}},"official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-from-delayed","slug":"reinforcement-learning-from-delayed","title":"Reinforcement Learning from Delayed Observations via World Models","date":"2024-03-18","arxiv_id":"2403.12309","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reinforcement-learning-from-delayed#ran","syntology_url":"https://syntology.ai/paper/2403.12309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12309"}},"official":{"repos":["indylab/delayeddreamer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/quality-diversity-actor-critic-learning-high","slug":"quality-diversity-actor-critic-learning-high","title":"Quality-Diversity Actor-Critic: Learning High-Performing and Diverse Behaviors via Value and Successor Features Critics","date":"2024-03-15","arxiv_id":"2403.09930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quality-diversity-actor-critic-learning-high#ran","syntology_url":"https://syntology.ai/paper/2403.09930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09930"}},"official":{"repos":["adaptive-intelligent-robotics/qdac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/splagger-split-aggregation-for-meta","slug":"splagger-split-aggregation-for-meta","title":"SplAgger: Split Aggregation for Meta-Reinforcement Learning","date":"2024-03-05","arxiv_id":"2403.03020","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/splagger-split-aggregation-for-meta#ran","syntology_url":"https://syntology.ai/paper/2403.03020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03020"}},"official":{"repos":["jacooba/hyper"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficientzero-v2-mastering-discrete-and","slug":"efficientzero-v2-mastering-discrete-and","title":"EfficientZero V2: Mastering Discrete and Continuous Control with Limited Data","date":"2024-03-01","arxiv_id":"2403.00564","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficientzero-v2-mastering-discrete-and#ran","syntology_url":"https://syntology.ai/paper/2403.00564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00564"}},"official":{"repos":["shengjiewang-jason/efficientzerov2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prise-learning-temporal-action-abstractions","slug":"prise-learning-temporal-action-abstractions","title":"PRISE: LLM-Style Sequence Compression for Learning Temporal Action Abstractions in Control","date":"2024-02-16","arxiv_id":"2402.10450","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prise-learning-temporal-action-abstractions#ran","syntology_url":"https://syntology.ai/paper/2402.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10450"}},"official":{"repos":["frankzheng2022/prise"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/premier-taco-pretraining-multitask","slug":"premier-taco-pretraining-multitask","title":"Premier-TACO is a Few-Shot Policy Learner: Pretraining Multitask Representation via Temporal Action-Driven Contrastive Loss","date":"2024-02-09","arxiv_id":"2402.06187","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/premier-taco-pretraining-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.06187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06187"}},"official":{"repos":["premiertaco/premier-taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reconciling-spatial-and-temporal-abstractions","slug":"reconciling-spatial-and-temporal-abstractions","title":"Reconciling Spatial and Temporal Abstractions for Goal Representation","date":"2024-01-18","arxiv_id":"2401.09870","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/reconciling-spatial-and-temporal-abstractions#ran","syntology_url":"https://syntology.ai/paper/2401.09870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09870"}},"official":{"repos":["cosynus-lix/STAR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/world-models-via-policy-guided-trajectory","slug":"world-models-via-policy-guided-trajectory","title":"World Models via Policy-Guided Trajectory Diffusion","date":"2023-12-13","arxiv_id":"2312.08533","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":2,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 2 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-models-via-policy-guided-trajectory#ran","syntology_url":"https://syntology.ai/paper/2312.08533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08533"}},"official":{"repos":["marc-rigter/polygrad-world-models"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decoupling-meta-reinforcement-learning-with","slug":"decoupling-meta-reinforcement-learning-with","title":"Decoupling Meta-Reinforcement Learning with Gaussian Task Contexts and Skills","date":"2023-12-11","arxiv_id":"2312.06518","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-meta-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2312.06518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06518"}},"official":{"repos":["hehongc/DCMRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/synergizing-quality-diversity-with-descriptor","slug":"synergizing-quality-diversity-with-descriptor","title":"Synergizing Quality-Diversity with Descriptor-Conditioned Reinforcement Learning","date":"2023-12-10","arxiv_id":"2401.08632","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/synergizing-quality-diversity-with-descriptor#ran","syntology_url":"https://syntology.ai/paper/2401.08632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08632"}},"official":{"repos":["adaptive-intelligent-robotics/DCRL-MAP-Elites"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/drm-mastering-visual-reinforcement-learning","slug":"drm-mastering-visual-reinforcement-learning","title":"DrM: Mastering Visual Reinforcement Learning through Dormant Ratio Minimization","date":"2023-10-30","arxiv_id":"2310.19668","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":3,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":6,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/drm-mastering-visual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.19668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19668"}},"official":{"repos":["XuGW-Kevin/DrM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/td-mpc2-scalable-robust-world-models-for","slug":"td-mpc2-scalable-robust-world-models-for","title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","date":"2023-10-25","arxiv_id":"2310.16828","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":4,"n_ran_checked":6,"n_instrument":6,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"12 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/td-mpc2-scalable-robust-world-models-for#ran","syntology_url":"https://syntology.ai/paper/2310.16828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16828"}},"official":null}},{"url":"/paper/absolute-policy-optimization","slug":"absolute-policy-optimization","title":"Absolute Policy Optimization","date":"2023-10-20","arxiv_id":"2310.13230","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/absolute-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2310.13230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13230"}},"official":{"repos":["intelligent-control-lab/absolute-policy-optimization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-continuous-control-with-consistency","slug":"boosting-continuous-control-with-consistency","title":"Boosting Continuous Control with Consistency Policy","date":"2023-10-10","arxiv_id":"2310.06343","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-continuous-control-with-consistency#ran","syntology_url":"https://syntology.ai/paper/2310.06343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06343"}},"official":{"repos":["cccedric/cpql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-optimization-in-a-noisy-neighborhood-1","slug":"policy-optimization-in-a-noisy-neighborhood-1","title":"Policy Optimization in a Noisy Neighborhood: On Return Landscapes in Continuous Control","date":"2023-09-26","arxiv_id":"2309.14597","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-optimization-in-a-noisy-neighborhood-1#ran","syntology_url":"https://syntology.ai/paper/2309.14597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14597"}},"official":{"repos":["nathanrahn/return-landscapes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/balancing-exploration-and-exploitation-in","slug":"balancing-exploration-and-exploitation-in","title":"Balancing Exploration and Exploitation in Hierarchical Reinforcement Learning via Latent Landmark Graphs","date":"2023-07-22","arxiv_id":"2307.12063","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/balancing-exploration-and-exploitation-in#ran","syntology_url":"https://syntology.ai/paper/2307.12063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12063"}},"official":{"repos":["papercode2022/hill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/taco-temporal-latent-action-driven","slug":"taco-temporal-latent-action-driven","title":"TACO: Temporal Latent Action-Driven Contrastive Loss for Visual Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.13229","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/taco-temporal-latent-action-driven#ran","syntology_url":"https://syntology.ai/paper/2306.13229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13229"}},"official":{"repos":["frankzheng2022/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/rosarl-reward-only-safe-reinforcement","slug":"rosarl-reward-only-safe-reinforcement","title":"ROSARL: Reward-Only Safe Reinforcement Learning","date":"2023-05-31","arxiv_id":"2306.00035","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rosarl-reward-only-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.00035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00035"}},"official":{"repos":["geraudnt/rosarl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/conditional-mutual-information-for-1","slug":"conditional-mutual-information-for-1","title":"Conditional Mutual Information for Disentangled Representations in Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-mutual-information-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.14133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14133"}},"official":{"repos":["uoe-agents/cmid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-contrastive-learning-for","slug":"behavior-contrastive-learning-for","title":"Behavior Contrastive Learning for Unsupervised Skill Discovery","date":"2023-05-08","arxiv_id":"2305.04477","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":1,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/behavior-contrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2305.04477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04477"}},"official":{"repos":["rooshy-yang/becl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-rl-decisions-with-trajectories","slug":"explaining-rl-decisions-with-trajectories","title":"Explaining RL Decisions with Trajectories","date":"2023-05-06","arxiv_id":"2305.04073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explaining-rl-decisions-with-trajectories#ran","syntology_url":"https://syntology.ai/paper/2305.04073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04073"}},"official":{"repos":["shripaddeshmukh/xrl_with_trajectories"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/masked-trajectory-models-for-prediction","slug":"masked-trajectory-models-for-prediction","title":"Masked Trajectory Models for Prediction, Representation, and Control","date":"2023-05-04","arxiv_id":"2305.02968","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-trajectory-models-for-prediction#ran","syntology_url":"https://syntology.ai/paper/2305.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02968"}},"official":{"repos":["facebookresearch/mtm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-state-abstraction-based-on","slug":"hierarchical-state-abstraction-based-on","title":"Hierarchical State Abstraction Based on Structural Information Principles","date":"2023-04-24","arxiv_id":"2304.12000","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hierarchical-state-abstraction-based-on#ran","syntology_url":"https://syntology.ai/paper/2304.12000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.12000"}},"official":{"repos":["ringbdstack/sisa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-state-space-models-for-in-context-1","slug":"structured-state-space-models-for-in-context-1","title":"Structured State Space Models for In-Context Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03982","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-state-space-models-for-in-context-1#ran","syntology_url":"https://syntology.ai/paper/2303.03982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03982"}},"official":{"repos":["luchris429/popjaxrl","luchris429/s5rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-decentralized-cross-entropy-method","slug":"a-simple-decentralized-cross-entropy-method","title":"A Simple Decentralized Cross-Entropy Method","date":"2022-12-16","arxiv_id":"2212.08235","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-simple-decentralized-cross-entropy-method#ran","syntology_url":"https://syntology.ai/paper/2212.08235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08235"}},"official":{"repos":["vincentzhang/decentcem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/q-pensieve-boosting-sample-efficiency-of","slug":"q-pensieve-boosting-sample-efficiency-of","title":"Q-Pensieve: Boosting Sample Efficiency of Multi-Objective RL Through Memory Sharing of Q-Snapshots","date":"2022-12-06","arxiv_id":"2212.03117","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/q-pensieve-boosting-sample-efficiency-of#ran","syntology_url":"https://syntology.ai/paper/2212.03117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03117"}},"official":{"repos":["NYCU-RL-Bandits-Lab/Q-Pensieve"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-crucial-is-transformer-in-decision","slug":"how-crucial-is-transformer-in-decision","title":"How Crucial is Transformer in Decision Transformer?","date":"2022-11-26","arxiv_id":"2211.14655","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/how-crucial-is-transformer-in-decision#ran","syntology_url":"https://syntology.ai/paper/2211.14655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14655"}},"official":{"repos":["max7born/decision-lstm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/erl-re-2-efficient-evolutionary-reinforcement","slug":"erl-re-2-efficient-evolutionary-reinforcement","title":"ERL-Re$^2$: Efficient Evolutionary Reinforcement Learning with Shared State Representation and Individual Policy Representation","date":"2022-10-26","arxiv_id":"2210.17375","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erl-re-2-efficient-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.17375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.17375"}},"official":{"repos":["yeshenpy/erl-re2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-evolution-of-linear-regions","slug":"understanding-the-evolution-of-linear-regions","title":"Understanding the Evolution of Linear Regions in Deep Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13611","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/understanding-the-evolution-of-linear-regions#ran","syntology_url":"https://syntology.ai/paper/2210.13611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13611"}},"official":{"repos":["setarehc/deep_rl_regions"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-continuous-control-via-q-learning","slug":"solving-continuous-control-via-q-learning","title":"Solving Continuous Control via Q-learning","date":"2022-10-22","arxiv_id":"2210.12566","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-continuous-control-via-q-learning#ran","syntology_url":"https://syntology.ai/paper/2210.12566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12566"}},"official":{"repos":["tseyde/decqn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-ask-for-help-proactive-interventions","slug":"when-to-ask-for-help-proactive-interventions","title":"When to Ask for Help: Proactive Interventions in Autonomous Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10765","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/when-to-ask-for-help-proactive-interventions#ran","syntology_url":"https://syntology.ai/paper/2210.10765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10765"}},"official":{"repos":["tajwarfahim/proactive_interventions"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-imitation-of-a-few-demonstrations-with","slug":"robust-imitation-of-a-few-demonstrations-with","title":"Robust Imitation of a Few Demonstrations with a Backwards Model","date":"2022-10-17","arxiv_id":"2210.09337","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-imitation-of-a-few-demonstrations-with#ran","syntology_url":"https://syntology.ai/paper/2210.09337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.09337"}},"official":null}},{"url":"/paper/a-unified-framework-for-alternating-offline","slug":"a-unified-framework-for-alternating-offline","title":"A Unified Framework for Alternating Offline Model Training and Policy Learning","date":"2022-10-12","arxiv_id":"2210.05922","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-alternating-offline#ran","syntology_url":"https://syntology.ai/paper/2210.05922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05922"}},"official":{"repos":["shentao-yang/ampl_neurips2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-state-marginalization-as-a-low-cost","slug":"latent-state-marginalization-as-a-low-cost","title":"Latent State Marginalization as a Low-cost Approach for Improving Exploration","date":"2022-10-03","arxiv_id":"2210.00999","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-state-marginalization-as-a-low-cost#ran","syntology_url":"https://syntology.ai/paper/2210.00999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00999"}},"official":{"repos":["zdhnarsil/stochastic-marginal-actor-critic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/continuous-mdp-homomorphisms-and-homomorphic","slug":"continuous-mdp-homomorphisms-and-homomorphic","title":"Continuous MDP Homomorphisms and Homomorphic Policy Gradient","date":"2022-09-15","arxiv_id":"2209.07364","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continuous-mdp-homomorphisms-and-homomorphic#ran","syntology_url":"https://syntology.ai/paper/2209.07364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.07364"}},"official":{"repos":["sahandrez/homomorphic_policy_gradient"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-planning-in-a-compact-latent-action","slug":"efficient-planning-in-a-compact-latent-action","title":"Efficient Planning in a Compact Latent Action Space","date":"2022-08-22","arxiv_id":"2208.10291","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-planning-in-a-compact-latent-action#ran","syntology_url":"https://syntology.ai/paper/2208.10291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10291"}},"official":{"repos":["ZhengyaoJiang/latentplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pd-morl-preference-driven-multi-objective","slug":"pd-morl-preference-driven-multi-objective","title":"PD-MORL: Preference-Driven Multi-Objective Reinforcement Learning Algorithm","date":"2022-08-16","arxiv_id":"2208.07914","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pd-morl-preference-driven-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2208.07914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.07914"}},"official":{"repos":["tbasaklar/PDMORL-Preference-Driven-Multi-Objective-Reinforcement-Learning-Algorithm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/live-in-the-moment-learning-dynamics-model","slug":"live-in-the-moment-learning-dynamics-model","title":"Live in the Moment: Learning Dynamics Model Adapted to Evolving Policy","date":"2022-07-25","arxiv_id":"2207.12141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/live-in-the-moment-learning-dynamics-model#ran","syntology_url":"https://syntology.ai/paper/2207.12141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12141"}},"official":{"repos":["si0wang/pdml"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/learning-bellman-complete-representations-for","slug":"learning-bellman-complete-representations-for","title":"Learning Bellman Complete Representations for Offline Policy Evaluation","date":"2022-07-12","arxiv_id":"2207.05837","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-bellman-complete-representations-for#ran","syntology_url":"https://syntology.ai/paper/2207.05837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05837"}},"official":{"repos":["causalml/bcrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-bandits-with-smooth-regret","slug":"contextual-bandits-with-smooth-regret","title":"Contextual Bandits with Smooth Regret: Efficient Learning in Continuous Action Spaces","date":"2022-07-12","arxiv_id":"2207.05849","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-bandits-with-smooth-regret#ran","syntology_url":"https://syntology.ai/paper/2207.05849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05849"}},"official":{"repos":["pmineiro/smoothcb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/general-policy-evaluation-and-improvement-by","slug":"general-policy-evaluation-and-improvement-by","title":"General Policy Evaluation and Improvement by Learning to Identify Few But Crucial States","date":"2022-07-04","arxiv_id":"2207.01566","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/general-policy-evaluation-and-improvement-by#ran","syntology_url":"https://syntology.ai/paper/2207.01566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01566"}},"official":{"repos":["idsia/policyevaluator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-conditioned-generators-of-deep-policies","slug":"goal-conditioned-generators-of-deep-policies","title":"Goal-Conditioned Generators of Deep Policies","date":"2022-07-04","arxiv_id":"2207.01570","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goal-conditioned-generators-of-deep-policies#ran","syntology_url":"https://syntology.ai/paper/2207.01570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01570"}},"official":{"repos":["idsia/gogepo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/transformers-are-meta-reinforcement-learners-1","slug":"transformers-are-meta-reinforcement-learners-1","title":"Transformers are Meta-Reinforcement Learners","date":"2022-06-14","arxiv_id":"2206.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transformers-are-meta-reinforcement-learners-1#ran","syntology_url":"https://syntology.ai/paper/2206.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06614"}},"official":{"repos":["luckeciano/transformers-metarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/defending-observation-attacks-in-deep","slug":"defending-observation-attacks-in-deep","title":"Defending Observation Attacks in Deep Reinforcement Learning via Detection and Denoising","date":"2022-06-14","arxiv_id":"2206.07188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/defending-observation-attacks-in-deep#ran","syntology_url":"https://syntology.ai/paper/2206.07188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07188"}},"official":{"repos":["ZikangXiong/rl-detect-and-denoise-defense"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-via-differentiable-physics","slug":"imitation-learning-via-differentiable-physics","title":"Imitation Learning via Differentiable Physics","date":"2022-06-10","arxiv_id":"2206.04873","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-via-differentiable-physics#ran","syntology_url":"https://syntology.ai/paper/2206.04873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04873"}},"official":{"repos":["sail-sg/ild"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tasil-taylor-series-imitation-learning","slug":"tasil-taylor-series-imitation-learning","title":"TaSIL: Taylor Series Imitation Learning","date":"2022-05-30","arxiv_id":"2205.14812","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tasil-taylor-series-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2205.14812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14812"}},"official":{"repos":["unstable-zeros/tasil"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-continual-reinforcement","slug":"task-agnostic-continual-reinforcement","title":"Task-Agnostic Continual Reinforcement Learning: Gaining Insights and Overcoming Challenges","date":"2022-05-28","arxiv_id":"2205.14495","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/task-agnostic-continual-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14495"}},"official":{"repos":["amazon-science/replay-based-recurrent-rl","amazon-research/replay-based-recurrent-rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/myosuite-a-contact-rich-simulation-suite-for","slug":"myosuite-a-contact-rich-simulation-suite-for","title":"MyoSuite -- A contact-rich simulation suite for musculoskeletal motor control","date":"2022-05-26","arxiv_id":"2205.13600","repositories_listed":3,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/myosuite-a-contact-rich-simulation-suite-for#ran","syntology_url":"https://syntology.ai/paper/2205.13600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13600"}},"official":{"repos":["facebookresearch/myosuite"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/skill-machines-temporal-logic-composition-in","slug":"skill-machines-temporal-logic-composition-in","title":"Skill Machines: Temporal Logic Skill Composition in Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12532","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/skill-machines-temporal-logic-composition-in#ran","syntology_url":"https://syntology.ai/paper/2205.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12532"}},"official":{"repos":["geraudnt/skill_machines"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-state-distribution-matching-approach-to-non","slug":"a-state-distribution-matching-approach-to-non","title":"A State-Distribution Matching Approach to Non-Episodic Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05212","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-state-distribution-matching-approach-to-non#ran","syntology_url":"https://syntology.ai/paper/2205.05212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05212"}},"official":{"repos":["architsharma97/medal"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/safe-reinforcement-learning-by-imagining-the-1","slug":"safe-reinforcement-learning-by-imagining-the-1","title":"Safe Reinforcement Learning by Imagining the Near Future","date":"2022-02-15","arxiv_id":"2202.07789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-by-imagining-the-1#ran","syntology_url":"https://syntology.ai/paper/2202.07789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.07789"}},"official":{"repos":["gwthomas/safe-mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-by-state-only-distribution","slug":"imitation-learning-by-state-only-distribution","title":"Imitation Learning by State-Only Distribution Matching","date":"2022-02-09","arxiv_id":"2202.04332","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-by-state-only-distribution#ran","syntology_url":"https://syntology.ai/paper/2202.04332","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04332"}},"official":{"repos":["FeMa42/soil-tdm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trusted-approximate-policy-iteration-with","slug":"trusted-approximate-policy-iteration-with","title":"Approximate Policy Iteration with Bisimulation Metrics","date":"2022-02-06","arxiv_id":"2202.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trusted-approximate-policy-iteration-with#ran","syntology_url":"https://syntology.ai/paper/2202.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02881"}},"official":{"repos":["metekemertas/api-bisim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarially-trained-actor-critic-for","slug":"adversarially-trained-actor-critic-for","title":"Adversarially Trained Actor Critic for Offline Reinforcement Learning","date":"2022-02-05","arxiv_id":"2202.02446","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarially-trained-actor-critic-for#ran","syntology_url":"https://syntology.ai/paper/2202.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02446"}},"official":{"repos":["microsoft/atac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/imitation-learning-by-estimating-expertise-of","slug":"imitation-learning-by-estimating-expertise-of","title":"Imitation Learning by Estimating Expertise of Demonstrators","date":"2022-02-02","arxiv_id":"2202.01288","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-by-estimating-expertise-of#ran","syntology_url":"https://syntology.ai/paper/2202.01288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01288"}},"official":{"repos":["stanford-iliad/ileed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dns-determinantal-point-process-based-neural","slug":"dns-determinantal-point-process-based-neural","title":"DNS: Determinantal Point Process Based Neural Network Sampler for Ensemble Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13357","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dns-determinantal-point-process-based-neural#ran","syntology_url":"https://syntology.ai/paper/2201.13357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13357"}},"official":{"repos":["IntelLabs/DNS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/sample-efficient-deep-reinforcement-learning-5","slug":"sample-efficient-deep-reinforcement-learning-5","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","date":"2022-01-05","arxiv_id":"2201.01666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-deep-reinforcement-learning-5#ran","syntology_url":"https://syntology.ai/paper/2201.01666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01666"}},"official":{"repos":["montrealrobotics/iv_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-experimental-design-perspective-on-model","slug":"an-experimental-design-perspective-on-model","title":"An Experimental Design Perspective on Model-Based Reinforcement Learning","date":"2021-12-09","arxiv_id":"2112.05244","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-experimental-design-perspective-on-model#ran","syntology_url":"https://syntology.ai/paper/2112.05244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05244"}},"official":null}},{"url":"/paper/calvin-a-benchmark-for-language-conditioned","slug":"calvin-a-benchmark-for-language-conditioned","title":"CALVIN: A Benchmark for Language-Conditioned Policy Learning for Long-Horizon Robot Manipulation Tasks","date":"2021-12-06","arxiv_id":"2112.03227","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/calvin-a-benchmark-for-language-conditioned#ran","syntology_url":"https://syntology.ai/paper/2112.03227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03227"}},"official":{"repos":["mees/calvin"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptively-calibrated-critic-estimates-for","slug":"adaptively-calibrated-critic-estimates-for","title":"Adaptively Calibrated Critic Estimates for Deep Reinforcement Learning","date":"2021-11-24","arxiv_id":"2111.12673","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptively-calibrated-critic-estimates-for#ran","syntology_url":"https://syntology.ai/paper/2111.12673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12673"}},"official":{"repos":["nicolinho/acc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-decision-transformer-for-offline","slug":"generalized-decision-transformer-for-offline","title":"Generalized Decision Transformer for Offline Hindsight Information Matching","date":"2021-11-19","arxiv_id":"2111.10364","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2111.10364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10364"}},"official":{"repos":["frt03/generalized_dt"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/curriculum-offline-imitation-learning","slug":"curriculum-offline-imitation-learning","title":"Curriculum Offline Imitation Learning","date":"2021-11-03","arxiv_id":"2111.02056","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curriculum-offline-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2111.02056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02056"}},"official":{"repos":["apexrl/coil"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/context-meta-reinforcement-learning-via","slug":"context-meta-reinforcement-learning-via","title":"Context Meta-Reinforcement Learning via Neuromodulation","date":"2021-10-30","arxiv_id":"2111.00134","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-meta-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2111.00134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.00134"}},"official":{"repos":["dlpbc/nm-metarl","soltoggio/ct-graph"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-skills-for-efficient-exploration","slug":"hierarchical-skills-for-efficient-exploration","title":"Hierarchical Skills for Efficient Exploration","date":"2021-10-20","arxiv_id":"2110.10809","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hierarchical-skills-for-efficient-exploration#ran","syntology_url":"https://syntology.ai/paper/2110.10809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.10809"}},"official":{"repos":["facebookresearch/hsd3"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-time-fitted-value-iteration-for","slug":"continuous-time-fitted-value-iteration-for","title":"Continuous-Time Fitted Value Iteration for Robust Policies","date":"2021-10-05","arxiv_id":"2110.01954","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continuous-time-fitted-value-iteration-for#ran","syntology_url":"https://syntology.ai/paper/2110.01954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01954"}},"official":null}},{"url":"/paper/imitation-learning-by-reinforcement-learning","slug":"imitation-learning-by-reinforcement-learning","title":"Imitation Learning by Reinforcement Learning","date":"2021-08-10","arxiv_id":"2108.04763","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-by-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2108.04763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.04763"}},"official":{"repos":["spotify-research/il-by-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mastering-visual-continuous-control-improved","slug":"mastering-visual-continuous-control-improved","title":"Mastering Visual Continuous Control: Improved Data-Augmented Reinforcement Learning","date":"2021-07-20","arxiv_id":"2107.09645","repositories_listed":8,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mastering-visual-continuous-control-improved#ran","syntology_url":"https://syntology.ai/paper/2107.09645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.09645"}},"official":{"repos":["facebookresearch/drqv2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/shortest-path-constrained-reinforcement-1","slug":"shortest-path-constrained-reinforcement-1","title":"Shortest-Path Constrained Reinforcement Learning for Sparse Reward Tasks","date":"2021-07-13","arxiv_id":"2107.06405","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shortest-path-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2107.06405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06405"}},"official":{"repos":["srsohn/shortest-path-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"d454feb0458fa0a797f5ca585a94c1784be571049cd7ed73e145d3045d34c3f5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}