{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/continuous-control/papers/3","list_of":"/task/continuous-control","task":"Continuous Control","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":12,"rows_per_page":100,"rows":[201,300],"of":1161,"counts":{"archive_papers_tagged":1161,"with_a_code_link":494,"where_syntology_ran_a_sample":193,"not_listed_spam_title":0,"listed":1161,"listed_where_code_ran":193,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":161,"every_run_a_failure_of_syntologys_instrument":32,"listed_with_a_run_with_no_instrument_failure":161,"listed_every_run_a_failure_of_syntologys_instrument":32,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/continuous-control","prev":"/task/continuous-control/papers/2","next":"/task/continuous-control/papers/4","papers":[{"url":"/paper/value-distributional-model-based","slug":"value-distributional-model-based","title":"Value-Distributional Model-Based Reinforcement Learning","date":"2023-08-12","arxiv_id":"2308.06590","repositories_listed":1,"syntology":null},{"url":"/paper/qdax-a-library-for-quality-diversity-and","slug":"qdax-a-library-for-quality-diversity-and","title":"QDax: A Library for Quality-Diversity and Population-based Algorithms with Hardware Acceleration","date":"2023-08-07","arxiv_id":"2308.03665","repositories_listed":1,"syntology":null},{"url":"/paper/balancing-exploration-and-exploitation-in","slug":"balancing-exploration-and-exploitation-in","title":"Balancing Exploration and Exploitation in Hierarchical Reinforcement Learning via Latent Landmark Graphs","date":"2023-07-22","arxiv_id":"2307.12063","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/balancing-exploration-and-exploitation-in#ran","syntology_url":"https://syntology.ai/paper/2307.12063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12063"}},"official":{"repos":["papercode2022/hill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-reinforcement-learning-techniques","slug":"exploring-reinforcement-learning-techniques","title":"Exploring reinforcement learning techniques for discrete and continuous control tasks in the MuJoCo environment","date":"2023-07-20","arxiv_id":"2307.11166","repositories_listed":1,"syntology":null},{"url":"/paper/taco-temporal-latent-action-driven","slug":"taco-temporal-latent-action-driven","title":"TACO: Temporal Latent Action-Driven Contrastive Loss for Visual Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.13229","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/taco-temporal-latent-action-driven#ran","syntology_url":"https://syntology.ai/paper/2306.13229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13229"}},"official":{"repos":["frankzheng2022/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","repositories_listed":1,"syntology":null},{"url":"/paper/seizing-serendipity-exploiting-the-value-of","slug":"seizing-serendipity-exploiting-the-value-of","title":"Seizing Serendipity: Exploiting the Value of Past Success in Off-Policy Actor-Critic","date":"2023-06-05","arxiv_id":"2306.02865","repositories_listed":1,"syntology":null},{"url":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","repositories_listed":1,"syntology":null},{"url":"/paper/rosarl-reward-only-safe-reinforcement","slug":"rosarl-reward-only-safe-reinforcement","title":"ROSARL: Reward-Only Safe Reinforcement Learning","date":"2023-05-31","arxiv_id":"2306.00035","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rosarl-reward-only-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.00035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00035"}},"official":{"repos":["geraudnt/rosarl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/temporally-layered-architecture-for-efficient","slug":"temporally-layered-architecture-for-efficient","title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","date":"2023-05-30","arxiv_id":"2305.18701","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-rl-algorithms-can-be-sample","slug":"off-policy-rl-algorithms-can-be-sample","title":"Off-Policy RL Algorithms Can be Sample-Efficient for Continuous Control via Sample Multiple Reuse","date":"2023-05-29","arxiv_id":"2305.18443","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-mutual-information-for-1","slug":"conditional-mutual-information-for-1","title":"Conditional Mutual Information for Disentangled Representations in Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-mutual-information-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.14133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14133"}},"official":{"repos":["uoe-agents/cmid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-contrastive-learning-for","slug":"behavior-contrastive-learning-for","title":"Behavior Contrastive Learning for Unsupervised Skill Discovery","date":"2023-05-08","arxiv_id":"2305.04477","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":1,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/behavior-contrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2305.04477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04477"}},"official":{"repos":["rooshy-yang/becl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-ensemble-directed-offline","slug":"federated-ensemble-directed-offline","title":"Federated Ensemble-Directed Offline Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.03097","repositories_listed":1,"syntology":null},{"url":"/paper/masked-trajectory-models-for-prediction","slug":"masked-trajectory-models-for-prediction","title":"Masked Trajectory Models for Prediction, Representation, and Control","date":"2023-05-04","arxiv_id":"2305.02968","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-trajectory-models-for-prediction#ran","syntology_url":"https://syntology.ai/paper/2305.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.02968"}},"official":{"repos":["facebookresearch/mtm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mixed-integer-optimal-control-via","slug":"mixed-integer-optimal-control-via","title":"Mixed-Integer Optimal Control via Reinforcement Learning: A Case Study on Hybrid Electric Vehicle Energy Management","date":"2023-05-02","arxiv_id":"2305.01461","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-state-abstraction-based-on","slug":"hierarchical-state-abstraction-based-on","title":"Hierarchical State Abstraction Based on Structural Information Principles","date":"2023-04-24","arxiv_id":"2304.12000","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hierarchical-state-abstraction-based-on#ran","syntology_url":"https://syntology.ai/paper/2304.12000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.12000"}},"official":{"repos":["ringbdstack/sisa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/uav-obstacle-avoidance-by-human-in-the-loop","slug":"uav-obstacle-avoidance-by-human-in-the-loop","title":"UAV Obstacle Avoidance by Human-in-the-Loop Reinforcement in Arbitrary 3D Environment","date":"2023-04-07","arxiv_id":"2304.05959","repositories_listed":1,"syntology":null},{"url":"/paper/on-context-distribution-shift-in-task","slug":"on-context-distribution-shift-in-task","title":"On Context Distribution Shift in Task Representation Learning for Offline Meta RL","date":"2023-04-01","arxiv_id":"2304.00354","repositories_listed":1,"syntology":null},{"url":"/paper/inverse-reinforcement-learning-without","slug":"inverse-reinforcement-learning-without","title":"Inverse Reinforcement Learning without Reinforcement Learning","date":"2023-03-26","arxiv_id":"2303.14623","repositories_listed":1,"syntology":null},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-real-time-planning-with","slug":"sample-efficient-real-time-planning-with","title":"Sample-efficient Real-time Planning with Curiosity Cross-Entropy Method and Contrastive Learning","date":"2023-03-07","arxiv_id":"2303.03787","repositories_listed":1,"syntology":null},{"url":"/paper/swim-a-general-purpose-high-performing-and","slug":"swim-a-general-purpose-high-performing-and","title":"Swim: A General-Purpose, High-Performing, and Efficient Activation Function for Locomotion Control Tasks","date":"2023-03-05","arxiv_id":"2303.02640","repositories_listed":1,"syntology":null},{"url":"/paper/hallucinated-adversarial-control-for","slug":"hallucinated-adversarial-control-for","title":"Hallucinated Adversarial Control for Conservative Offline Policy Evaluation","date":"2023-03-02","arxiv_id":"2303.01076","repositories_listed":1,"syntology":null},{"url":"/paper/ls-iq-implicit-reward-regularization-for","slug":"ls-iq-implicit-reward-regularization-for","title":"LS-IQ: Implicit Reward Regularization for Inverse Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00599","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-descriptor-based-control-for-deep","slug":"continuous-descriptor-based-control-for-deep","title":"Continuous descriptor-based control for deep audio synthesis","date":"2023-02-27","arxiv_id":"2302.13542","repositories_listed":1,"syntology":null},{"url":"/paper/crystalbox-future-based-explanations-for-drl","slug":"crystalbox-future-based-explanations-for-drl","title":"CrystalBox: Future-Based Explanations for Input-Driven Deep RL Systems","date":"2023-02-27","arxiv_id":"2302.13483","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-morphology-control-via-contextual","slug":"universal-morphology-control-via-contextual","title":"Universal Morphology Control via Contextual Modulation","date":"2023-02-22","arxiv_id":"2302.11070","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-cooperative-multi-agent","slug":"attacking-cooperative-multi-agent","title":"Attacking Cooperative Multi-Agent Reinforcement Learning by Adversarial Minority Influence","date":"2023-02-07","arxiv_id":"2302.03322","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-transport-perturbations-for-safe","slug":"optimal-transport-perturbations-for-safe","title":"Optimal Transport Perturbations for Safe Reinforcement Learning with Robustness Guarantees","date":"2023-01-31","arxiv_id":"2301.13375","repositories_listed":1,"syntology":null},{"url":"/paper/centralized-cooperative-exploration-policy","slug":"centralized-cooperative-exploration-policy","title":"Centralized Cooperative Exploration Policy for Continuous Control Tasks","date":"2023-01-06","arxiv_id":"2301.02375","repositories_listed":1,"syntology":null},{"url":"/paper/learning-goal-conditioned-policies-offline","slug":"learning-goal-conditioned-policies-offline","title":"Learning Goal-Conditioned Policies Offline with Self-Supervised Reward Shaping","date":"2023-01-05","arxiv_id":"2301.02099","repositories_listed":1,"syntology":null},{"url":"/paper/robust-control-for-dynamical-systems-with-non","slug":"robust-control-for-dynamical-systems-with-non","title":"Robust Control for Dynamical Systems With Non-Gaussian Noise via Formal Abstractions","date":"2023-01-04","arxiv_id":"2301.01526","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-decentralized-cross-entropy-method","slug":"a-simple-decentralized-cross-entropy-method","title":"A Simple Decentralized Cross-Entropy Method","date":"2022-12-16","arxiv_id":"2212.08235","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-simple-decentralized-cross-entropy-method#ran","syntology_url":"https://syntology.ai/paper/2212.08235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08235"}},"official":{"repos":["vincentzhang/decentcem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/q-pensieve-boosting-sample-efficiency-of","slug":"q-pensieve-boosting-sample-efficiency-of","title":"Q-Pensieve: Boosting Sample Efficiency of Multi-Objective RL Through Memory Sharing of Q-Snapshots","date":"2022-12-06","arxiv_id":"2212.03117","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/q-pensieve-boosting-sample-efficiency-of#ran","syntology_url":"https://syntology.ai/paper/2212.03117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03117"}},"official":{"repos":["NYCU-RL-Bandits-Lab/Q-Pensieve"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/variable-decision-frequency-option-critic","slug":"variable-decision-frequency-option-critic","title":"Dynamic Decision Frequency with Continuous Options","date":"2022-12-06","arxiv_id":"2212.04407","repositories_listed":1,"syntology":null},{"url":"/paper/policy-learning-for-active-target-tracking","slug":"policy-learning-for-active-target-tracking","title":"Policy Learning for Active Target Tracking over Continuous SE(3) Trajectories","date":"2022-12-03","arxiv_id":"2212.01498","repositories_listed":1,"syntology":null},{"url":"/paper/stl-based-synthesis-of-feedback-controllers","slug":"stl-based-synthesis-of-feedback-controllers","title":"STL-Based Synthesis of Feedback Controllers Using Reinforcement Learning","date":"2022-12-02","arxiv_id":"2212.01022","repositories_listed":1,"syntology":null},{"url":"/paper/how-crucial-is-transformer-in-decision","slug":"how-crucial-is-transformer-in-decision","title":"How Crucial is Transformer in Decision Transformer?","date":"2022-11-26","arxiv_id":"2211.14655","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/how-crucial-is-transformer-in-decision#ran","syntology_url":"https://syntology.ai/paper/2211.14655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14655"}},"official":{"repos":["max7born/decision-lstm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/a-system-for-morphology-task-generalization","slug":"a-system-for-morphology-task-generalization","title":"A System for Morphology-Task Generalization via Unified Representation and Behavior Distillation","date":"2022-11-25","arxiv_id":"2211.14296","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-evolution-strategies-via-meta","slug":"discovering-evolution-strategies-via-meta","title":"Discovering Evolution Strategies via Meta-Black-Box Optimization","date":"2022-11-21","arxiv_id":"2211.11260","repositories_listed":1,"syntology":null},{"url":"/paper/fi-ode-certified-and-robust-forward","slug":"fi-ode-certified-and-robust-forward","title":"FI-ODE: Certifiably Robust Forward Invariance in Neural ODEs","date":"2022-10-30","arxiv_id":"2210.16940","repositories_listed":1,"syntology":null},{"url":"/paper/erl-re-2-efficient-evolutionary-reinforcement","slug":"erl-re-2-efficient-evolutionary-reinforcement","title":"ERL-Re$^2$: Efficient Evolutionary Reinforcement Learning with Shared State Representation and Individual Policy Representation","date":"2022-10-26","arxiv_id":"2210.17375","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erl-re-2-efficient-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2210.17375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.17375"}},"official":{"repos":["yeshenpy/erl-re2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-evolution-of-linear-regions","slug":"understanding-the-evolution-of-linear-regions","title":"Understanding the Evolution of Linear Regions in Deep Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13611","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/understanding-the-evolution-of-linear-regions#ran","syntology_url":"https://syntology.ai/paper/2210.13611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13611"}},"official":{"repos":["setarehc/deep_rl_regions"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-continuous-control-via-q-learning","slug":"solving-continuous-control-via-q-learning","title":"Solving Continuous Control via Q-learning","date":"2022-10-22","arxiv_id":"2210.12566","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-continuous-control-via-q-learning#ran","syntology_url":"https://syntology.ai/paper/2210.12566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12566"}},"official":{"repos":["tseyde/decqn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rate-splitting-for-intelligent-reflecting","slug":"rate-splitting-for-intelligent-reflecting","title":"Rate-Splitting for Intelligent Reflecting Surface-Aided Multiuser VR Streaming","date":"2022-10-21","arxiv_id":"2210.12191","repositories_listed":1,"syntology":null},{"url":"/paper/when-to-ask-for-help-proactive-interventions","slug":"when-to-ask-for-help-proactive-interventions","title":"When to Ask for Help: Proactive Interventions in Autonomous Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10765","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/when-to-ask-for-help-proactive-interventions#ran","syntology_url":"https://syntology.ai/paper/2210.10765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10765"}},"official":{"repos":["tajwarfahim/proactive_interventions"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-alternating-offline","slug":"a-unified-framework-for-alternating-offline","title":"A Unified Framework for Alternating Offline Model Training and Policy Learning","date":"2022-10-12","arxiv_id":"2210.05922","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-alternating-offline#ran","syntology_url":"https://syntology.ai/paper/2210.05922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05922"}},"official":{"repos":["shentao-yang/ampl_neurips2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-control-policies-for-region","slug":"learning-control-policies-for-region","title":"Learning Provably Stabilizing Neural Controllers for Discrete-Time Stochastic Systems","date":"2022-10-11","arxiv_id":"2210.05304","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-bayesian-model-based-value","slug":"conservative-bayesian-model-based-value","title":"Conservative Bayesian Model-Based Value Expansion for Offline Policy Optimization","date":"2022-10-07","arxiv_id":"2210.03802","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-monte-carlo-graph-search","slug":"continuous-monte-carlo-graph-search","title":"Continuous Monte Carlo Graph Search","date":"2022-10-04","arxiv_id":"2210.01426","repositories_listed":1,"syntology":null},{"url":"/paper/latent-state-marginalization-as-a-low-cost","slug":"latent-state-marginalization-as-a-low-cost","title":"Latent State Marginalization as a Low-cost Approach for Improving Exploration","date":"2022-10-03","arxiv_id":"2210.00999","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-state-marginalization-as-a-low-cost#ran","syntology_url":"https://syntology.ai/paper/2210.00999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00999"}},"official":{"repos":["zdhnarsil/stochastic-marginal-actor-critic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/dmap-a-distributed-morphological-attention","slug":"dmap-a-distributed-morphological-attention","title":"DMAP: a Distributed Morphological Attention Policy for Learning to Locomote with a Changing Body","date":"2022-09-28","arxiv_id":"2209.14218","repositories_listed":1,"syntology":null},{"url":"/paper/learning-continuous-control-policies-for","slug":"learning-continuous-control-policies-for","title":"Learning Continuous Control Policies for Information-Theoretic Active Perception","date":"2022-09-26","arxiv_id":"2209.12427","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-mdp-homomorphisms-and-homomorphic","slug":"continuous-mdp-homomorphisms-and-homomorphic","title":"Continuous MDP Homomorphisms and Homomorphic Policy Gradient","date":"2022-09-15","arxiv_id":"2209.07364","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continuous-mdp-homomorphisms-and-homomorphic#ran","syntology_url":"https://syntology.ai/paper/2209.07364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.07364"}},"official":{"repos":["sahandrez/homomorphic_policy_gradient"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploiting-reward-shifting-in-value-based","slug":"exploiting-reward-shifting-in-value-based","title":"Optimistic Curiosity Exploration and Conservative Exploitation with Linear Reward Shaping","date":"2022-09-15","arxiv_id":"2209.07288","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-reuse-bias-in-off-policy-reinforcement","slug":"on-the-reuse-bias-in-off-policy-reinforcement","title":"On the Reuse Bias in Off-Policy Reinforcement Learning","date":"2022-09-15","arxiv_id":"2209.07074","repositories_listed":1,"syntology":null},{"url":"/paper/forlorn-a-framework-for-comparing-offline","slug":"forlorn-a-framework-for-comparing-offline","title":"FORLORN: A Framework for Comparing Offline Methods and Reinforcement Learning for Optimization of RAN Parameters","date":"2022-09-08","arxiv_id":"2209.13540","repositories_listed":1,"syntology":null},{"url":"/paper/actor-prioritized-experience-replay","slug":"actor-prioritized-experience-replay","title":"Actor Prioritized Experience Replay","date":"2022-09-01","arxiv_id":"2209.00532","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-planning-in-a-compact-latent-action","slug":"efficient-planning-in-a-compact-latent-action","title":"Efficient Planning in a Compact Latent Action Space","date":"2022-08-22","arxiv_id":"2208.10291","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-planning-in-a-compact-latent-action#ran","syntology_url":"https://syntology.ai/paper/2208.10291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.10291"}},"official":{"repos":["ZhengyaoJiang/latentplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pd-morl-preference-driven-multi-objective","slug":"pd-morl-preference-driven-multi-objective","title":"PD-MORL: Preference-Driven Multi-Objective Reinforcement Learning Algorithm","date":"2022-08-16","arxiv_id":"2208.07914","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pd-morl-preference-driven-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2208.07914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.07914"}},"official":{"repos":["tbasaklar/PDMORL-Preference-Driven-Multi-Objective-Reinforcement-Learning-Algorithm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sequence-model-imitation-learning-with","slug":"sequence-model-imitation-learning-with","title":"Sequence Model Imitation Learning with Unobserved Contexts","date":"2022-08-03","arxiv_id":"2208.02225","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-correction-for-actor-critic","slug":"off-policy-correction-for-actor-critic","title":"Mitigating Off-Policy Bias in Actor-Critic Methods with One-Step Q-learning: A Novel Correction Approach","date":"2022-08-01","arxiv_id":"2208.00755","repositories_listed":1,"syntology":null},{"url":"/paper/safe-and-robust-experience-sharing-for","slug":"safe-and-robust-experience-sharing-for","title":"Safe and Robust Experience Sharing for Deterministic Policy Gradient Algorithms","date":"2022-07-27","arxiv_id":"2207.13453","repositories_listed":1,"syntology":null},{"url":"/paper/live-in-the-moment-learning-dynamics-model","slug":"live-in-the-moment-learning-dynamics-model","title":"Live in the Moment: Learning Dynamics Model Adapted to Evolving Policy","date":"2022-07-25","arxiv_id":"2207.12141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/live-in-the-moment-learning-dynamics-model#ran","syntology_url":"https://syntology.ai/paper/2207.12141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12141"}},"official":{"repos":["si0wang/pdml"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/contextual-bandits-with-smooth-regret","slug":"contextual-bandits-with-smooth-regret","title":"Contextual Bandits with Smooth Regret: Efficient Learning in Continuous Action Spaces","date":"2022-07-12","arxiv_id":"2207.05849","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-bandits-with-smooth-regret#ran","syntology_url":"https://syntology.ai/paper/2207.05849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05849"}},"official":{"repos":["pmineiro/smoothcb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-bellman-complete-representations-for","slug":"learning-bellman-complete-representations-for","title":"Learning Bellman Complete Representations for Offline Policy Evaluation","date":"2022-07-12","arxiv_id":"2207.05837","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/learning-bellman-complete-representations-for#ran","syntology_url":"https://syntology.ai/paper/2207.05837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05837"}},"official":{"repos":["causalml/bcrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-in-continuous","slug":"robust-reinforcement-learning-in-continuous","title":"Robust Reinforcement Learning in Continuous Control Tasks with Uncertainty Set Regularization","date":"2022-07-05","arxiv_id":"2207.02016","repositories_listed":1,"syntology":null},{"url":"/paper/general-policy-evaluation-and-improvement-by","slug":"general-policy-evaluation-and-improvement-by","title":"General Policy Evaluation and Improvement by Learning to Identify Few But Crucial States","date":"2022-07-04","arxiv_id":"2207.01566","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/general-policy-evaluation-and-improvement-by#ran","syntology_url":"https://syntology.ai/paper/2207.01566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01566"}},"official":{"repos":["idsia/policyevaluator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-policy-optimization-with-eligible","slug":"offline-policy-optimization-with-eligible","title":"Offline Policy Optimization with Eligible Actions","date":"2022-07-01","arxiv_id":"2207.00632","repositories_listed":1,"syntology":null},{"url":"/paper/guided-exploration-in-reinforcement-learning","slug":"guided-exploration-in-reinforcement-learning","title":"Guided Exploration in Reinforcement Learning via Monte Carlo Critic Optimization","date":"2022-06-25","arxiv_id":"2206.12674","repositories_listed":1,"syntology":null},{"url":"/paper/defending-observation-attacks-in-deep","slug":"defending-observation-attacks-in-deep","title":"Defending Observation Attacks in Deep Reinforcement Learning via Detection and Denoising","date":"2022-06-14","arxiv_id":"2206.07188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/defending-observation-attacks-in-deep#ran","syntology_url":"https://syntology.ai/paper/2206.07188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07188"}},"official":{"repos":["ZikangXiong/rl-detect-and-denoise-defense"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-meta-reinforcement-learners-1","slug":"transformers-are-meta-reinforcement-learners-1","title":"Transformers are Meta-Reinforcement Learners","date":"2022-06-14","arxiv_id":"2206.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transformers-are-meta-reinforcement-learners-1#ran","syntology_url":"https://syntology.ai/paper/2206.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06614"}},"official":{"repos":["luckeciano/transformers-metarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-via-differentiable-physics","slug":"imitation-learning-via-differentiable-physics","title":"Imitation Learning via Differentiable Physics","date":"2022-06-10","arxiv_id":"2206.04873","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-via-differentiable-physics#ran","syntology_url":"https://syntology.ai/paper/2206.04873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04873"}},"official":{"repos":["sail-sg/ild"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-safe-reinforcement-learning-via-2","slug":"towards-safe-reinforcement-learning-via-2","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value-at-Risk","date":"2022-06-09","arxiv_id":"2206.04436","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-optimal-online-imitation-learning-via","slug":"minimax-optimal-online-imitation-learning-via","title":"Minimax Optimal Online Imitation Learning via Replay Estimation","date":"2022-05-30","arxiv_id":"2205.15397","repositories_listed":1,"syntology":null},{"url":"/paper/rlx2-training-a-sparse-deep-reinforcement","slug":"rlx2-training-a-sparse-deep-reinforcement","title":"RLx2: Training a Sparse Deep Reinforcement Learning Model from Scratch","date":"2022-05-30","arxiv_id":"2205.15043","repositories_listed":1,"syntology":null},{"url":"/paper/tasil-taylor-series-imitation-learning","slug":"tasil-taylor-series-imitation-learning","title":"TaSIL: Taylor Series Imitation Learning","date":"2022-05-30","arxiv_id":"2205.14812","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tasil-taylor-series-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2205.14812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14812"}},"official":{"repos":["unstable-zeros/tasil"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/skill-machines-temporal-logic-composition-in","slug":"skill-machines-temporal-logic-composition-in","title":"Skill Machines: Temporal Logic Skill Composition in Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12532","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/skill-machines-temporal-logic-composition-in#ran","syntology_url":"https://syntology.ai/paper/2205.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12532"}},"official":{"repos":["geraudnt/skill_machines"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/neighborhood-mixup-experience-replay-local","slug":"neighborhood-mixup-experience-replay-local","title":"Neighborhood Mixup Experience Replay: Local Convex Interpolation for Improved Sample Efficiency in Continuous Control Tasks","date":"2022-05-18","arxiv_id":"2205.09117","repositories_listed":1,"syntology":null},{"url":"/paper/a-state-distribution-matching-approach-to-non","slug":"a-state-distribution-matching-approach-to-non","title":"A State-Distribution Matching Approach to Non-Episodic Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05212","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-state-distribution-matching-approach-to-non#ran","syntology_url":"https://syntology.ai/paper/2205.05212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05212"}},"official":{"repos":["architsharma97/medal"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simultaneous-double-q-learning-with","slug":"simultaneous-double-q-learning-with","title":"Simultaneous Double Q-learning with Conservative Advantage Learning for Actor-Critic Methods","date":"2022-05-08","arxiv_id":"2205.03819","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-gaussian-mixture-critic-in-off","slug":"revisiting-gaussian-mixture-critic-in-off","title":"Revisiting Gaussian mixture critics in off-policy reinforcement learning: a sample-based approach","date":"2022-04-21","arxiv_id":"2204.10256","repositories_listed":1,"syntology":null},{"url":"/paper/automating-reinforcement-learning-with","slug":"automating-reinforcement-learning-with","title":"Automating Reinforcement Learning with Example-based Resets","date":"2022-04-05","arxiv_id":"2204.02041","repositories_listed":1,"syntology":null},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/safe-reinforcement-learning-by-imagining-the-1","slug":"safe-reinforcement-learning-by-imagining-the-1","title":"Safe Reinforcement Learning by Imagining the Near Future","date":"2022-02-15","arxiv_id":"2202.07789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-by-imagining-the-1#ran","syntology_url":"https://syntology.ai/paper/2202.07789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.07789"}},"official":{"repos":["gwthomas/safe-mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-learning-from-observation-with-model","slug":"robust-learning-from-observation-with-model","title":"Robust Learning from Observation with Model Misspecification","date":"2022-02-12","arxiv_id":"2202.06003","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-by-state-only-distribution","slug":"imitation-learning-by-state-only-distribution","title":"Imitation Learning by State-Only Distribution Matching","date":"2022-02-09","arxiv_id":"2202.04332","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-by-state-only-distribution#ran","syntology_url":"https://syntology.ai/paper/2202.04332","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04332"}},"official":{"repos":["FeMa42/soil-tdm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bingham-policy-parameterization-for-3d","slug":"bingham-policy-parameterization-for-3d","title":"Bingham Policy Parameterization for 3D Rotations in Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03957","repositories_listed":1,"syntology":null},{"url":"/paper/trusted-approximate-policy-iteration-with","slug":"trusted-approximate-policy-iteration-with","title":"Approximate Policy Iteration with Bisimulation Metrics","date":"2022-02-06","arxiv_id":"2202.02881","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trusted-approximate-policy-iteration-with#ran","syntology_url":"https://syntology.ai/paper/2202.02881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02881"}},"official":{"repos":["metekemertas/api-bisim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-interpretable-high-performing","slug":"learning-interpretable-high-performing","title":"Learning Interpretable, High-Performing Policies for Autonomous Driving","date":"2022-02-04","arxiv_id":"2202.02352","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-by-estimating-expertise-of","slug":"imitation-learning-by-estimating-expertise-of","title":"Imitation Learning by Estimating Expertise of Demonstrators","date":"2022-02-02","arxiv_id":"2202.01288","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-by-estimating-expertise-of#ran","syntology_url":"https://syntology.ai/paper/2202.01288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01288"}},"official":{"repos":["stanford-iliad/ileed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dns-determinantal-point-process-based-neural","slug":"dns-determinantal-point-process-based-neural","title":"DNS: Determinantal Point Process Based Neural Network Sampler for Ensemble Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13357","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dns-determinantal-point-process-based-neural#ran","syntology_url":"https://syntology.ai/paper/2201.13357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13357"}},"official":{"repos":["IntelLabs/DNS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/zeroth-order-actor-critic-1","slug":"zeroth-order-actor-critic-1","title":"Zeroth-Order Actor-Critic: An Evolutionary Framework for Sequential Decision Problems","date":"2022-01-29","arxiv_id":"2201.12518","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-5","slug":"sample-efficient-deep-reinforcement-learning-5","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","date":"2022-01-05","arxiv_id":"2201.01666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-deep-reinforcement-learning-5#ran","syntology_url":"https://syntology.ai/paper/2201.01666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01666"}},"official":{"repos":["montrealrobotics/iv_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-causal-aware-rl-state-wise-action","slug":"toward-causal-aware-rl-state-wise-action","title":"Toward Causal-Aware RL: State-Wise Action-Refined Temporal Difference","date":"2022-01-02","arxiv_id":"2201.00354","repositories_listed":1,"syntology":null}],"record_sha256":"109f21900932e3155c35a7b23e636546c9c8cf7a136469e95f7e51025e06f8b5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}