{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/continuous-control/papers/2","list_of":"/task/continuous-control","task":"Continuous Control","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":12,"rows_per_page":100,"rows":[101,200],"of":1161,"counts":{"archive_papers_tagged":1161,"with_a_code_link":494,"where_syntology_ran_a_sample":193,"not_listed_spam_title":0,"listed":1161,"listed_where_code_ran":193,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/continuous-control","prev":"/task/continuous-control","next":"/task/continuous-control/papers/3","papers":[{"url":"/paper/regularization-matters-in-policy-optimization-1","slug":"regularization-matters-in-policy-optimization-1","title":"Regularization Matters in Policy Optimization","date":"2019-10-21","arxiv_id":"1910.09191","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularization-matters-in-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/1910.09191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09191"}},"official":{"repos":["xuanlinli17/iclr2021_rlreg","xuanlinli17/po-rl-regularization"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/meta-q-learning","slug":"meta-q-learning","title":"Meta-Q-Learning","date":"2019-09-30","arxiv_id":"1910.00125","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-q-learning#ran","syntology_url":"https://syntology.ai/paper/1910.00125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00125"}},"official":{"repos":["amazon-research/meta-q-learning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/driving-in-dense-traffic-with-model-free","slug":"driving-in-dense-traffic-with-model-free","title":"Driving in Dense Traffic with Model-Free Reinforcement Learning","date":"2019-09-15","arxiv_id":"1909.06710","repositories_listed":2,"syntology":null},{"url":"/paper/dynamics-aware-embeddings","slug":"dynamics-aware-embeddings","title":"Dynamics-aware Embeddings","date":"2019-08-25","arxiv_id":"1908.09357","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dynamics-aware-embeddings#ran","syntology_url":"https://syntology.ai/paper/1908.09357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.09357"}},"official":{"repos":["dyne-submission/dynamics-aware-embeddings"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/cobra-data-efficient-model-based-rl-through","slug":"cobra-data-efficient-model-based-rl-through","title":"COBRA: Data-Efficient Model-Based RL through Unsupervised Object Discovery and Curiosity-Driven Exploration","date":"2019-05-22","arxiv_id":"1905.09275","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cobra-data-efficient-model-based-rl-through#ran","syntology_url":"https://syntology.ai/paper/1905.09275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09275"}},"official":{"repos":["deepmind/spriteworld"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-based-off-policy-deep","slug":"trajectory-based-off-policy-deep","title":"Trajectory-Based Off-Policy Deep Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05710","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-feedback","slug":"deep-reinforcement-learning-with-feedback","title":"Deep Reinforcement Learning with Feedback-based Exploration","date":"2019-03-14","arxiv_id":"1903.06151","repositories_listed":2,"syntology":null},{"url":"/paper/discretizing-continuous-action-space-for-on","slug":"discretizing-continuous-action-space-for-on","title":"Discretizing Continuous Action Space for On-Policy Optimization","date":"2019-01-29","arxiv_id":"1901.10500","repositories_listed":2,"syntology":null},{"url":"/paper/action-robust-reinforcement-learning-and","slug":"action-robust-reinforcement-learning-and","title":"Action Robust Reinforcement Learning and Applications in Continuous Control","date":"2019-01-26","arxiv_id":"1901.09184","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-robust-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/1901.09184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09184"}},"official":{"repos":["tesslerc/ActionRobustRL","icml2019-anonymous-author/Action-Robust-Reinforcement-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-reinforcement-learning","slug":"benchmarking-reinforcement-learning","title":"Benchmarking Reinforcement Learning Algorithms on Real-World Robots","date":"2018-09-20","arxiv_id":"1809.07731","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1809.07731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.07731"}},"official":{"repos":["kindredresearch/SenseAct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-optimization-via-importance-sampling","slug":"policy-optimization-via-importance-sampling","title":"Policy Optimization via Importance Sampling","date":"2018-09-17","arxiv_id":"1809.06098","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-optimization-via-importance-sampling#ran","syntology_url":"https://syntology.ai/paper/1809.06098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06098"}},"official":{"repos":["T3p/pois"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/algorithmic-framework-for-model-based-deep","slug":"algorithmic-framework-for-model-based-deep","title":"Algorithmic Framework for Model-based Deep Reinforcement Learning with Theoretical Guarantees","date":"2018-07-10","arxiv_id":"1807.03858","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/algorithmic-framework-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1807.03858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03858"}},"official":{"repos":["roosephu/slbo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-reinforcement-learning-with","slug":"sample-efficient-reinforcement-learning-with","title":"Sample-Efficient Reinforcement Learning with Stochastic Ensemble Value Expansion","date":"2018-07-04","arxiv_id":"1807.01675","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-adapt-in-dynamic-real-world","slug":"learning-to-adapt-in-dynamic-real-world","title":"Learning to Adapt in Dynamic, Real-World Environments Through Meta-Reinforcement Learning","date":"2018-03-30","arxiv_id":"1803.11347","repositories_listed":2,"syntology":null},{"url":"/paper/model-ensemble-trust-region-policy","slug":"model-ensemble-trust-region-policy","title":"Model-Ensemble Trust-Region Policy Optimization","date":"2018-02-28","arxiv_id":"1802.10592","repositories_listed":2,"syntology":{"n":6,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 6 unverified","sample_list":"/paper/model-ensemble-trust-region-policy#ran","syntology_url":"https://syntology.ai/paper/1802.10592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10592"}},"official":{"repos":["thanard/me-trpo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":[]}}},{"url":"/paper/learning-model-based-planning-from-scratch","slug":"learning-model-based-planning-from-scratch","title":"Learning model-based planning from scratch","date":"2017-07-19","arxiv_id":"1707.06170","repositories_listed":2,"syntology":null},{"url":"/paper/q-prop-sample-efficient-policy-gradient-with","slug":"q-prop-sample-efficient-policy-gradient-with","title":"Q-Prop: Sample-Efficient Policy Gradient with An Off-Policy Critic","date":"2016-11-07","arxiv_id":"1611.02247","repositories_listed":2,"syntology":null},{"url":"/paper/modular-multitask-reinforcement-learning-with","slug":"modular-multitask-reinforcement-learning-with","title":"Modular Multitask Reinforcement Learning with Policy Sketches","date":"2016-11-06","arxiv_id":"1611.01796","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-multitask-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1611.01796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01796"}},"official":{"repos":["jacobandreas/psketch"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vime-variational-information-maximizing","slug":"vime-variational-information-maximizing","title":"VIME: Variational Information Maximizing Exploration","date":"2016-05-31","arxiv_id":"1605.09674","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vime-variational-information-maximizing#ran","syntology_url":"https://syntology.ai/paper/1605.09674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1605.09674"}},"official":null}},{"url":"/paper/sparse-reg-improving-sample-complexity-in","slug":"sparse-reg-improving-sample-complexity-in","title":"Sparse-Reg: Improving Sample Complexity in Offline Reinforcement Learning using Sparsity","date":"2025-06-20","arxiv_id":"2506.17155","repositories_listed":1,"syntology":null},{"url":"/paper/dr-sac-distributionally-robust-soft-actor","slug":"dr-sac-distributionally-robust-soft-actor","title":"DR-SAC: Distributionally Robust Soft Actor-Critic for Reinforcement Learning under Uncertainty","date":"2025-06-14","arxiv_id":"2506.12622","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/dr-sac-distributionally-robust-soft-actor#ran","syntology_url":"https://syntology.ai/paper/2506.12622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12622"}},"official":{"repos":["lemutisme/dr-sac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-value-estimation-critically","slug":"improving-value-estimation-critically","title":"Improving Value Estimation Critically Enhances Vanilla Policy Gradient","date":"2025-05-25","arxiv_id":"2505.19247","repositories_listed":1,"syntology":null},{"url":"/paper/guided-policy-optimization-under-partial","slug":"guided-policy-optimization-under-partial","title":"Guided Policy Optimization under Partial Observability","date":"2025-05-21","arxiv_id":"2505.15418","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/guided-policy-optimization-under-partial#ran","syntology_url":"https://syntology.ai/paper/2505.15418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15418"}},"official":{"repos":["liyheng/GPO"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/rlbenchnet-the-right-network-for-the-right","slug":"rlbenchnet-the-right-network-for-the-right","title":"RLBenchNet: The Right Network for the Right Reinforcement Learning Task","date":"2025-05-21","arxiv_id":"2505.15040","repositories_listed":1,"syntology":null},{"url":"/paper/sample-and-computationally-efficient-1","slug":"sample-and-computationally-efficient-1","title":"Sample and Computationally Efficient Continuous-Time Reinforcement Learning with General Function Approximation","date":"2025-05-20","arxiv_id":"2505.14821","repositories_listed":1,"syntology":null},{"url":"/paper/cie-controlling-language-model-text","slug":"cie-controlling-language-model-text","title":"CIE: Controlling Language Model Text Generations Using Continuous Signals","date":"2025-05-19","arxiv_id":"2505.13448","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-robust-tracking-control-an-online","slug":"enhanced-robust-tracking-control-an-online","title":"Enhanced Robust Tracking Control: An Online Learning Approach","date":"2025-05-08","arxiv_id":"2505.05036","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-llms-in-human-in-the-loop-rl","slug":"zero-shot-llms-in-human-in-the-loop-rl","title":"Zero-Shot LLMs in Human-in-the-Loop RL: Replacing Human Feedback for Reward Shaping","date":"2025-03-26","arxiv_id":"2503.22723","repositories_listed":1,"syntology":null},{"url":"/paper/learning-with-expert-abstractions-for","slug":"learning-with-expert-abstractions-for","title":"Learning with Expert Abstractions for Efficient Multi-Task Continuous Control","date":"2025-03-19","arxiv_id":"2503.14809","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-policy-optimization-for-offline","slug":"adversarial-policy-optimization-for-offline","title":"Adversarial Policy Optimization for Offline Preference-based Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-policy-optimization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2503.05306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05306"}},"official":{"repos":["oh-lab/APPO"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/discrete-codebook-world-models-for-continuous","slug":"discrete-codebook-world-models-for-continuous","title":"Discrete Codebook World Models for Continuous Control","date":"2025-03-01","arxiv_id":"2503.00653","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discrete-codebook-world-models-for-continuous#ran","syntology_url":"https://syntology.ai/paper/2503.00653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00653"}},"official":{"repos":["aidanscannell/dcmpc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-decision-making-in-stochastic","slug":"scalable-decision-making-in-stochastic","title":"Scalable Decision-Making in Stochastic Environments through Learned Temporal Abstraction","date":"2025-02-28","arxiv_id":"2502.21186","repositories_listed":1,"syntology":null},{"url":"/paper/langevin-soft-actor-critic-efficient","slug":"langevin-soft-actor-critic-efficient","title":"Langevin Soft Actor-Critic: Efficient Exploration through Uncertainty-Driven Critic Learning","date":"2025-01-29","arxiv_id":"2501.17827","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/langevin-soft-actor-critic-efficient#ran","syntology_url":"https://syntology.ai/paper/2501.17827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17827"}},"official":{"repos":["hmishfaq/lsac"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pixelbrax-learning-continuous-control-from","slug":"pixelbrax-learning-continuous-control-from","title":"PixelBrax: Learning Continuous Control from Pixels End-to-End on the GPU","date":"2025-01-16","arxiv_id":"2502.00021","repositories_listed":1,"syntology":null},{"url":"/paper/smose-sparse-mixture-of-shallow-experts-for","slug":"smose-sparse-mixture-of-shallow-experts-for","title":"SMOSE: Sparse Mixture of Shallow Experts for Interpretable Reinforcement Learning in Continuous Control Tasks","date":"2024-12-17","arxiv_id":"2412.13053","repositories_listed":1,"syntology":null},{"url":"/paper/sinergym-a-virtual-testbed-for-building","slug":"sinergym-a-virtual-testbed-for-building","title":"SINERGYM -- A virtual testbed for building energy optimization with Reinforcement Learning","date":"2024-12-11","arxiv_id":"2412.08293","repositories_listed":1,"syntology":null},{"url":"/paper/meta-controller-few-shot-imitation-of-unseen","slug":"meta-controller-few-shot-imitation-of-unseen","title":"Meta-Controller: Few-Shot Imitation of Unseen Embodiments and Tasks in Continuous Control","date":"2024-12-10","arxiv_id":"2412.12147","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meta-controller-few-shot-imitation-of-unseen#ran","syntology_url":"https://syntology.ai/paper/2412.12147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12147"}},"official":{"repos":["seongwoongcho/meta-controller"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-based-model-predictive-control-for-5","slug":"learning-based-model-predictive-control-for-5","title":"Learning-Based Model Predictive Control for Piecewise Affine Systems with Feasibility Guarantees","date":"2024-11-30","arxiv_id":"2412.00490","repositories_listed":1,"syntology":null},{"url":"/paper/object-centric-proto-symbolic-behavioural","slug":"object-centric-proto-symbolic-behavioural","title":"Object-centric proto-symbolic behavioural reasoning from pixels","date":"2024-11-26","arxiv_id":"2411.17438","repositories_listed":1,"syntology":null},{"url":"/paper/cale-continuous-arcade-learning-environment","slug":"cale-continuous-arcade-learning-environment","title":"CALE: Continuous Arcade Learning Environment","date":"2024-10-31","arxiv_id":"2410.23810","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cale-continuous-arcade-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2410.23810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23810"}},"official":{"repos":["farama-foundation/arcade-learning-environment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusing-states-and-matching-scores-a-new","slug":"diffusing-states-and-matching-scores-a-new","title":"Diffusing States and Matching Scores: A New Framework for Imitation Learning","date":"2024-10-17","arxiv_id":"2410.13855","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffusing-states-and-matching-scores-a-new#ran","syntology_url":"https://syntology.ai/paper/2410.13855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13855"}},"official":{"repos":["ziqian2000/smiling"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/orso-accelerating-reward-design-via-online","slug":"orso-accelerating-reward-design-via-online","title":"ORSO: Accelerating Reward Design via Online Reward Selection and Policy Optimization","date":"2024-10-17","arxiv_id":"2410.13837","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-euclidean-data","slug":"reinforcement-learning-with-euclidean-data","title":"Reinforcement Learning with Euclidean Data Augmentation for State-Based Continuous Control","date":"2024-10-16","arxiv_id":"2410.12983","repositories_listed":1,"syntology":null},{"url":"/paper/overcoming-slow-decision-frequencies-in","slug":"overcoming-slow-decision-frequencies-in","title":"Overcoming Slow Decision Frequencies in Continuous Control: Model-Based Sequence Reinforcement Learning for Model-Free Control","date":"2024-10-11","arxiv_id":"2410.08979","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":3,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/overcoming-slow-decision-frequencies-in#ran","syntology_url":"https://syntology.ai/paper/2410.08979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08979"}},"official":{"repos":["dee0512/Temporally-Layered-Architecture"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/bisimulation-metric-for-model-predictive","slug":"bisimulation-metric-for-model-predictive","title":"Bisimulation metric for Model Predictive Control","date":"2024-10-06","arxiv_id":"2410.04553","repositories_listed":1,"syntology":null},{"url":"/paper/c-morl-multi-objective-reinforcement-learning","slug":"c-morl-multi-objective-reinforcement-learning","title":"C-MORL: Multi-Objective Reinforcement Learning through Efficient Discovery of Pareto Front","date":"2024-10-03","arxiv_id":"2410.02236","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/c-morl-multi-objective-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2410.02236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02236"}},"official":{"repos":["ruohliuq/c-morl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dmc-vb-a-benchmark-for-representation","slug":"dmc-vb-a-benchmark-for-representation","title":"DMC-VB: A Benchmark for Representation Learning for Control with Visual Distractors","date":"2024-09-26","arxiv_id":"2409.18330","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dmc-vb-a-benchmark-for-representation#ran","syntology_url":"https://syntology.ai/paper/2409.18330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18330"}},"official":{"repos":["google-deepmind/dmc_vision_benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/soft-actor-critic-with-beta-policy-via","slug":"soft-actor-critic-with-beta-policy-via","title":"Soft Actor-Critic with Beta Policy via Implicit Reparameterization Gradients","date":"2024-09-08","arxiv_id":"2409.04971","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-policy-policy-optimization","slug":"diffusion-policy-policy-optimization","title":"Diffusion Policy Policy Optimization","date":"2024-09-01","arxiv_id":"2409.00588","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-policy-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2409.00588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00588"}},"official":null}},{"url":"/paper/multi-agent-continuous-control-with","slug":"multi-agent-continuous-control-with","title":"Multi-Agent Continuous Control with Generative Flow Networks","date":"2024-08-13","arxiv_id":"2408.06920","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-transfer-learning-for-contextual","slug":"model-based-transfer-learning-for-contextual","title":"Model-Based Transfer Learning for Contextual Reinforcement Learning","date":"2024-08-08","arxiv_id":"2408.04498","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-transfer-learning-for-contextual#ran","syntology_url":"https://syntology.ai/paper/2408.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04498"}},"official":{"repos":["jhoon-cho/mbtl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-gaussian-temporal-difference","slug":"generalized-gaussian-temporal-difference","title":"Generalized Gaussian Temporal Difference Error for Uncertainty-aware Reinforcement Learning","date":"2024-08-05","arxiv_id":"2408.02295","repositories_listed":1,"syntology":null},{"url":"/paper/discretizing-continuous-action-space-with","slug":"discretizing-continuous-action-space-with","title":"Discretizing Continuous Action Space with Unimodal Probability Distributions for On-Policy Reinforcement Learning","date":"2024-08-01","arxiv_id":"2408.00309","repositories_listed":1,"syntology":null},{"url":"/paper/black-box-meta-learning-intrinsic-rewards-for","slug":"black-box-meta-learning-intrinsic-rewards-for","title":"Black box meta-learning intrinsic rewards for sparse-reward environments","date":"2024-07-31","arxiv_id":"2407.21546","repositories_listed":1,"syntology":null},{"url":"/paper/proximal-policy-distillation","slug":"proximal-policy-distillation","title":"Proximal Policy Distillation","date":"2024-07-21","arxiv_id":"2407.15134","repositories_listed":1,"syntology":null},{"url":"/paper/aligning-diffusion-behaviors-with-q-functions","slug":"aligning-diffusion-behaviors-with-q-functions","title":"Aligning Diffusion Behaviors with Q-functions for Efficient Continuous Control","date":"2024-07-12","arxiv_id":"2407.09024","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/aligning-diffusion-behaviors-with-q-functions#ran","syntology_url":"https://syntology.ai/paper/2407.09024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09024"}},"official":{"repos":["thu-ml/efficient-diffusion-alignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/roer-regularized-optimal-experience-replay","slug":"roer-regularized-optimal-experience-replay","title":"ROER: Regularized Optimal Experience Replay","date":"2024-07-04","arxiv_id":"2407.03995","repositories_listed":1,"syntology":null},{"url":"/paper/robocupgym-a-challenging-continuous-control","slug":"robocupgym-a-challenging-continuous-control","title":"RobocupGym: A challenging continuous control benchmark in Robocup","date":"2024-07-03","arxiv_id":"2407.14516","repositories_listed":1,"syntology":null},{"url":"/paper/behaviour-distillation","slug":"behaviour-distillation","title":"Behaviour Distillation","date":"2024-06-21","arxiv_id":"2406.15042","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behaviour-distillation#ran","syntology_url":"https://syntology.ai/paper/2406.15042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15042"}},"official":{"repos":["flairox/behaviour-distillation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-minimal-reinforcement-learning","slug":"discovering-minimal-reinforcement-learning","title":"Discovering Minimal Reinforcement Learning Environments","date":"2024-06-18","arxiv_id":"2406.12589","repositories_listed":1,"syntology":null},{"url":"/paper/evil-evolution-strategies-for-generalisable","slug":"evil-evolution-strategies-for-generalisable","title":"EvIL: Evolution Strategies for Generalisable Imitation Learning","date":"2024-06-15","arxiv_id":"2406.11905","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evil-evolution-strategies-for-generalisable#ran","syntology_url":"https://syntology.ai/paper/2406.11905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11905"}},"official":{"repos":["SilviaSapora/evil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rrls-robust-reinforcement-learning-suite","slug":"rrls-robust-reinforcement-learning-suite","title":"RRLS : Robust Reinforcement Learning Suite","date":"2024-06-12","arxiv_id":"2406.08406","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rrls-robust-reinforcement-learning-suite#ran","syntology_url":"https://syntology.ai/paper/2406.08406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08406"}},"official":{"repos":["sureli/rrls"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plandq-hierarchical-plan-orchestration-via-d","slug":"plandq-hierarchical-plan-orchestration-via-d","title":"PlanDQ: Hierarchical Plan Orchestration via D-Conductor and Q-Performer","date":"2024-06-10","arxiv_id":"2406.06793","repositories_listed":1,"syntology":null},{"url":"/paper/amortizing-intractable-inference-in-diffusion","slug":"amortizing-intractable-inference-in-diffusion","title":"Amortizing intractable inference in diffusion models for vision, language, and control","date":"2024-05-31","arxiv_id":"2405.20971","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/amortizing-intractable-inference-in-diffusion#ran","syntology_url":"https://syntology.ai/paper/2405.20971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20971"}},"official":{"repos":["gfnorg/diffusion-finetuning"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gaussian-flow-bridges-for-audio-domain","slug":"gaussian-flow-bridges-for-audio-domain","title":"Gaussian Flow Bridges for Audio Domain Transfer with Unpaired Data","date":"2024-05-29","arxiv_id":"2405.19497","repositories_listed":1,"syntology":null},{"url":"/paper/bigger-regularized-optimistic-scaling-for","slug":"bigger-regularized-optimistic-scaling-for","title":"Bigger, Regularized, Optimistic: scaling for compute and sample-efficient continuous control","date":"2024-05-25","arxiv_id":"2405.16158","repositories_listed":1,"syntology":{"n":20,"n_ran":11,"n_constructed":7,"n_ran_checked":10,"n_instrument":1,"n_unverified":9,"n_honours":2,"n_violates":1,"n_no_contract":7,"n_pointer_only":3,"phrase":"11 ran (of which 7 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/bigger-regularized-optimistic-scaling-for#ran","syntology_url":"https://syntology.ai/paper/2405.16158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16158"}},"official":{"repos":["naumix/BiggerRegularizedOptimistic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/diffusion-based-reinforcement-learning-via-q","slug":"diffusion-based-reinforcement-learning-via-q","title":"Diffusion-based Reinforcement Learning via Q-weighted Variational Policy Optimization","date":"2024-05-25","arxiv_id":"2405.16173","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-leverage-diverse-demonstrations-in","slug":"how-to-leverage-diverse-demonstrations-in","title":"How to Leverage Diverse Demonstrations in Offline Imitation Learning","date":"2024-05-24","arxiv_id":"2405.17476","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/how-to-leverage-diverse-demonstrations-in#ran","syntology_url":"https://syntology.ai/paper/2405.17476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17476"}},"official":{"repos":["hansenhua/ilid-offline-imitation-learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/ollie-imitation-learning-from-offline","slug":"ollie-imitation-learning-from-offline","title":"OLLIE: Imitation Learning from Offline Pretraining to Online Finetuning","date":"2024-05-24","arxiv_id":"2405.17477","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":4,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/ollie-imitation-learning-from-offline#ran","syntology_url":"https://syntology.ai/paper/2405.17477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17477"}},"official":{"repos":["hansenhua/ollie-offline-to-online-imitation-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/offline-reinforcement-learning-from-datasets","slug":"offline-reinforcement-learning-from-datasets","title":"Offline Reinforcement Learning from Datasets with Structured Non-Stationarity","date":"2024-05-23","arxiv_id":"2405.14114","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/offline-reinforcement-learning-from-datasets#ran","syntology_url":"https://syntology.ai/paper/2405.14114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14114"}},"official":{"repos":["johannesack/offlinerlstructurednonstationarity"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ctd4-a-deep-continuous-distributional-actor","slug":"ctd4-a-deep-continuous-distributional-actor","title":"CTD4 -- A Deep Continuous Distributional Actor-Critic Agent with a Kalman Fusion of Multiple Critics","date":"2024-05-04","arxiv_id":"2405.02576","repositories_listed":1,"syntology":null},{"url":"/paper/afu-actor-free-critic-updates-in-off-policy","slug":"afu-actor-free-critic-updates-in-off-policy","title":"AFU: Actor-Free critic Updates in off-policy RL for continuous control","date":"2024-04-24","arxiv_id":"2404.16159","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-regularization-of-representation","slug":"adaptive-regularization-of-representation","title":"Adaptive Regularization of Representation Rank as an Implicit Constraint of Bellman Equation","date":"2024-04-19","arxiv_id":"2404.12754","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-from-delayed","slug":"reinforcement-learning-from-delayed","title":"Reinforcement Learning from Delayed Observations via World Models","date":"2024-03-18","arxiv_id":"2403.12309","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reinforcement-learning-from-delayed#ran","syntology_url":"https://syntology.ai/paper/2403.12309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12309"}},"official":{"repos":["indylab/delayeddreamer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/quality-diversity-actor-critic-learning-high","slug":"quality-diversity-actor-critic-learning-high","title":"Quality-Diversity Actor-Critic: Learning High-Performing and Diverse Behaviors via Value and Successor Features Critics","date":"2024-03-15","arxiv_id":"2403.09930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quality-diversity-actor-critic-learning-high#ran","syntology_url":"https://syntology.ai/paper/2403.09930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09930"}},"official":{"repos":["adaptive-intelligent-robotics/qdac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/splagger-split-aggregation-for-meta","slug":"splagger-split-aggregation-for-meta","title":"SplAgger: Split Aggregation for Meta-Reinforcement Learning","date":"2024-03-05","arxiv_id":"2403.03020","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/splagger-split-aggregation-for-meta#ran","syntology_url":"https://syntology.ai/paper/2403.03020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03020"}},"official":{"repos":["jacooba/hyper"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficientzero-v2-mastering-discrete-and","slug":"efficientzero-v2-mastering-discrete-and","title":"EfficientZero V2: Mastering Discrete and Continuous Control with Limited Data","date":"2024-03-01","arxiv_id":"2403.00564","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficientzero-v2-mastering-discrete-and#ran","syntology_url":"https://syntology.ai/paper/2403.00564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00564"}},"official":{"repos":["shengjiewang-jason/efficientzerov2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-model-based-approach-for-improving","slug":"a-model-based-approach-for-improving","title":"A Model-Based Approach for Improving Reinforcement Learning Efficiency Leveraging Expert Observations","date":"2024-02-29","arxiv_id":"2402.18836","repositories_listed":1,"syntology":null},{"url":"/paper/prise-learning-temporal-action-abstractions","slug":"prise-learning-temporal-action-abstractions","title":"PRISE: LLM-Style Sequence Compression for Learning Temporal Action Abstractions in Control","date":"2024-02-16","arxiv_id":"2402.10450","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prise-learning-temporal-action-abstractions#ran","syntology_url":"https://syntology.ai/paper/2402.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10450"}},"official":{"repos":["frankzheng2022/prise"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dataset-clustering-for-improved-offline","slug":"dataset-clustering-for-improved-offline","title":"Dataset Clustering for Improved Offline Policy Learning","date":"2024-02-14","arxiv_id":"2402.09550","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-inverse-reinforcement-learning","slug":"hybrid-inverse-reinforcement-learning","title":"Hybrid Inverse Reinforcement Learning","date":"2024-02-13","arxiv_id":"2402.08848","repositories_listed":1,"syntology":null},{"url":"/paper/premier-taco-pretraining-multitask","slug":"premier-taco-pretraining-multitask","title":"Premier-TACO is a Few-Shot Policy Learner: Pretraining Multitask Representation via Temporal Action-Driven Contrastive Loss","date":"2024-02-09","arxiv_id":"2402.06187","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/premier-taco-pretraining-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.06187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06187"}},"official":{"repos":["premiertaco/premier-taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flowpg-action-constrained-policy-gradient-1","slug":"flowpg-action-constrained-policy-gradient-1","title":"FlowPG: Action-constrained Policy Gradient with Normalizing Flows","date":"2024-02-07","arxiv_id":"2402.05149","repositories_listed":1,"syntology":null},{"url":"/paper/logical-specifications-guided-dynamic-task","slug":"logical-specifications-guided-dynamic-task","title":"Logical Specifications-guided Dynamic Task Sampling for Reinforcement Learning Agents","date":"2024-02-06","arxiv_id":"2402.03678","repositories_listed":1,"syntology":null},{"url":"/paper/the-definitive-guide-to-policy-gradients-in","slug":"the-definitive-guide-to-policy-gradients-in","title":"The Definitive Guide to Policy Gradients in Deep Reinforcement Learning: Theory, Algorithms and Implementations","date":"2024-01-24","arxiv_id":"2401.13662","repositories_listed":1,"syntology":null},{"url":"/paper/reconciling-spatial-and-temporal-abstractions","slug":"reconciling-spatial-and-temporal-abstractions","title":"Reconciling Spatial and Temporal Abstractions for Goal Representation","date":"2024-01-18","arxiv_id":"2401.09870","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/reconciling-spatial-and-temporal-abstractions#ran","syntology_url":"https://syntology.ai/paper/2401.09870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09870"}},"official":{"repos":["cosynus-lix/STAR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/the-distributional-reward-critic-architecture","slug":"the-distributional-reward-critic-architecture","title":"The Distributional Reward Critic Framework for Reinforcement Learning Under Perturbed Rewards","date":"2024-01-11","arxiv_id":"2401.05710","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-based-interactive-imitation-learning","slug":"ensemble-based-interactive-imitation-learning","title":"Agnostic Interactive Imitation Learning: New Theory and Practical Algorithms","date":"2023-12-28","arxiv_id":"2312.16860","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-analysis-of-policy-networks-an","slug":"generalization-analysis-of-policy-networks-an","title":"Analyzing Generalization in Policy Networks: A Case Study with the Double-Integrator System","date":"2023-12-16","arxiv_id":"2312.10472","repositories_listed":1,"syntology":null},{"url":"/paper/risk-aware-continuous-control-with-neural","slug":"risk-aware-continuous-control-with-neural","title":"Risk-Aware Continuous Control with Neural Contextual Bandits","date":"2023-12-15","arxiv_id":"2312.09961","repositories_listed":1,"syntology":null},{"url":"/paper/world-models-via-policy-guided-trajectory","slug":"world-models-via-policy-guided-trajectory","title":"World Models via Policy-Guided Trajectory Diffusion","date":"2023-12-13","arxiv_id":"2312.08533","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":2,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 2 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-models-via-policy-guided-trajectory#ran","syntology_url":"https://syntology.ai/paper/2312.08533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08533"}},"official":{"repos":["marc-rigter/polygrad-world-models"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decoupling-meta-reinforcement-learning-with","slug":"decoupling-meta-reinforcement-learning-with","title":"Decoupling Meta-Reinforcement Learning with Gaussian Task Contexts and Skills","date":"2023-12-11","arxiv_id":"2312.06518","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-meta-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2312.06518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06518"}},"official":{"repos":["hehongc/DCMRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/absolute-policy-optimization","slug":"absolute-policy-optimization","title":"Absolute Policy Optimization","date":"2023-10-20","arxiv_id":"2310.13230","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/absolute-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2310.13230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13230"}},"official":{"repos":["intelligent-control-lab/absolute-policy-optimization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reduced-policy-optimization-for-continuous","slug":"reduced-policy-optimization-for-continuous","title":"Reduced Policy Optimization for Continuous Control with Hard Constraints","date":"2023-10-14","arxiv_id":"2310.09574","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-continuous-control-with-consistency","slug":"boosting-continuous-control-with-consistency","title":"Boosting Continuous Control with Consistency Policy","date":"2023-10-10","arxiv_id":"2310.06343","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-continuous-control-with-consistency#ran","syntology_url":"https://syntology.ai/paper/2310.06343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06343"}},"official":{"repos":["cccedric/cpql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-optimization-in-a-noisy-neighborhood-1","slug":"policy-optimization-in-a-noisy-neighborhood-1","title":"Policy Optimization in a Noisy Neighborhood: On Return Landscapes in Continuous Control","date":"2023-09-26","arxiv_id":"2309.14597","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-optimization-in-a-noisy-neighborhood-1#ran","syntology_url":"https://syntology.ai/paper/2309.14597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14597"}},"official":{"repos":["nathanrahn/return-landscapes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-shared-safety-constraints-from-multi-1","slug":"learning-shared-safety-constraints-from-multi-1","title":"Learning Shared Safety Constraints from Multi-task Demonstrations","date":"2023-09-01","arxiv_id":"2309.00711","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-unsupervised-environment-design","slug":"stabilizing-unsupervised-environment-design","title":"Stabilizing Unsupervised Environment Design with a Learned Adversary","date":"2023-08-21","arxiv_id":"2308.10797","repositories_listed":1,"syntology":null},{"url":"/paper/acre-actor-critic-with-reward-preserving","slug":"acre-actor-critic-with-reward-preserving","title":"ACRE: Actor-Critic with Reward-Preserving Exploration","date":"2023-08-14","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"9d4d8f0fb9548133828da8031fbb98b72deb83fb1518920f6c4fe5bbb5ec749d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}