{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mujoco/papers/2","list_of":"/task/mujoco","task":"MuJoCo","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":7,"rows_per_page":100,"rows":[101,200],"of":677,"counts":{"archive_papers_tagged":677,"with_a_code_link":293,"where_syntology_ran_a_sample":115,"not_listed_spam_title":0,"listed":677,"listed_where_code_ran":115,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":103,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":103,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mujoco","prev":"/task/mujoco","next":"/task/mujoco/papers/3","papers":[{"url":"/paper/variational-delayed-policy-optimization","slug":"variational-delayed-policy-optimization","title":"Variational Delayed Policy Optimization","date":"2024-05-23","arxiv_id":"2405.14226","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/variational-delayed-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2405.14226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14226"}},"official":{"repos":["qingyuanwunothing/vdpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/maximum-entropy-reinforcement-learning-via","slug":"maximum-entropy-reinforcement-learning-via","title":"Maximum Entropy Reinforcement Learning via Energy-Based Normalizing Flow","date":"2024-05-22","arxiv_id":"2405.13629","repositories_listed":1,"syntology":null},{"url":"/paper/is-mamba-compatible-with-trajectory","slug":"is-mamba-compatible-with-trajectory","title":"Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning?","date":"2024-05-20","arxiv_id":"2405.12094","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-mamba-compatible-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2405.12094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12094"}},"official":{"repos":["AndssY/DeMa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-with-1","slug":"robust-deep-reinforcement-learning-with-1","title":"Robust Deep Reinforcement Learning with Adaptive Adversarial Perturbations in Action Space","date":"2024-05-20","arxiv_id":"2405.11982","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2405.11982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11982"}},"official":{"repos":["lqm00/a2p-sac"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-exploration-for-data-efficient","slug":"adaptive-exploration-for-data-efficient","title":"Adaptive Exploration for Data-Efficient General Value Function Evaluations","date":"2024-05-13","arxiv_id":"2405.07838","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/adaptive-exploration-for-data-efficient#ran","syntology_url":"https://syntology.ai/paper/2405.07838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07838"}},"official":{"repos":["arushijain94/explorationofgvfs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hard-thresholding-meets-evolution-strategies","slug":"hard-thresholding-meets-evolution-strategies","title":"Hard-Thresholding Meets Evolution Strategies in Reinforcement Learning","date":"2024-05-02","arxiv_id":"2405.01615","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hard-thresholding-meets-evolution-strategies#ran","syntology_url":"https://syntology.ai/paper/2405.01615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01615"}},"official":{"repos":["cangcn/nes-ht"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/s-2-ac-energy-based-reinforcement-learning","slug":"s-2-ac-energy-based-reinforcement-learning","title":"S$^2$AC: Energy-Based Reinforcement Learning with Stein Soft Actor Critic","date":"2024-05-02","arxiv_id":"2405.00987","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/s-2-ac-energy-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2405.00987","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.00987"}},"official":{"repos":["safamessaoud/s2ac-energy-based-rl-with-stein-soft-actor-critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/no-representation-no-trust-connecting","slug":"no-representation-no-trust-connecting","title":"No Representation, No Trust: Connecting Representation, Collapse, and Trust Issues in PPO","date":"2024-05-01","arxiv_id":"2405.00662","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/no-representation-no-trust-connecting#ran","syntology_url":"https://syntology.ai/paper/2405.00662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.00662"}},"official":{"repos":["claire-labo/no-representation-no-trust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ucb-driven-utility-function-search-for-multi","slug":"ucb-driven-utility-function-search-for-multi","title":"UCB-driven Utility Function Search for Multi-objective Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.00410","repositories_listed":1,"syntology":null},{"url":"/paper/humanoid-gym-reinforcement-learning-for","slug":"humanoid-gym-reinforcement-learning-for","title":"Humanoid-Gym: Reinforcement Learning for Humanoid Robot with Zero-Shot Sim2Real Transfer","date":"2024-04-08","arxiv_id":"2404.05695","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humanoid-gym-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2404.05695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05695"}},"official":{"repos":["roboterax/humanoid-gym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/snapshot-reinforcement-learning-leveraging","slug":"snapshot-reinforcement-learning-leveraging","title":"Snapshot Reinforcement Learning: Leveraging Prior Trajectories for Efficiency","date":"2024-03-01","arxiv_id":"2403.00673","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-worst-case-attacks-robust-rl-with","slug":"beyond-worst-case-attacks-robust-rl-with","title":"Beyond Worst-case Attacks: Robust RL with Adaptive Defense via Non-dominated Policies","date":"2024-02-20","arxiv_id":"2402.12673","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/beyond-worst-case-attacks-robust-rl-with#ran","syntology_url":"https://syntology.ai/paper/2402.12673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12673"}},"official":{"repos":["umd-huang-lab/protected"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/debiased-offline-representation-learning-for","slug":"debiased-offline-representation-learning-for","title":"Debiased Offline Representation Learning for Fast Online Adaptation in Non-stationary Dynamics","date":"2024-02-17","arxiv_id":"2402.11317","repositories_listed":1,"syntology":null},{"url":"/paper/latent-plan-transformer-planning-as-latent","slug":"latent-plan-transformer-planning-as-latent","title":"Latent Plan Transformer for Trajectory Abstraction: Planning as Latent Space Inference","date":"2024-02-07","arxiv_id":"2402.04647","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-plan-transformer-planning-as-latent#ran","syntology_url":"https://syntology.ai/paper/2402.04647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04647"}},"official":{"repos":["mingluzhao/latent-plan-transformer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/expert-proximity-as-surrogate-rewards-for","slug":"expert-proximity-as-surrogate-rewards-for","title":"Expert Proximity as Surrogate Rewards for Single Demonstration Imitation Learning","date":"2024-02-01","arxiv_id":"2402.01057","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/expert-proximity-as-surrogate-rewards-for#ran","syntology_url":"https://syntology.ai/paper/2402.01057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01057"}},"official":{"repos":["stanl1y/tdil"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simple-policy-optimization","slug":"simple-policy-optimization","title":"Simple Policy Optimization","date":"2024-01-29","arxiv_id":"2401.16025","repositories_listed":1,"syntology":null},{"url":"/paper/an-invariant-information-geometric-method-for","slug":"an-invariant-information-geometric-method-for","title":"An Invariant Information Geometric Method for High-Dimensional Online Optimization","date":"2024-01-03","arxiv_id":"2401.01579","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-trajectory-constrained-exploration","slug":"adaptive-trajectory-constrained-exploration","title":"Adaptive trajectory-constrained exploration strategy for deep reinforcement learning","date":"2023-12-27","arxiv_id":"2312.16456","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-and-pessimistic-actor-in-rl","slug":"optimistic-and-pessimistic-actor-in-rl","title":"Efficient Reinforcement Learning via Decoupling Exploration and Utilization","date":"2023-12-26","arxiv_id":"2312.15965","repositories_listed":1,"syntology":null},{"url":"/paper/xuance-a-comprehensive-and-unified-deep","slug":"xuance-a-comprehensive-and-unified-deep","title":"XuanCe: A Comprehensive and Unified Deep Reinforcement Learning Library","date":"2023-12-25","arxiv_id":"2312.16248","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/xuance-a-comprehensive-and-unified-deep#ran","syntology_url":"https://syntology.ai/paper/2312.16248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16248"}},"official":{"repos":["agi-brain/xuance"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/go-dice-goal-conditioned-option-aware-offline","slug":"go-dice-goal-conditioned-option-aware-offline","title":"GO-DICE: Goal-Conditioned Option-Aware Offline Imitation Learning via Stationary Distribution Correction Estimation","date":"2023-12-17","arxiv_id":"2312.10802","repositories_listed":1,"syntology":null},{"url":"/paper/world-models-via-policy-guided-trajectory","slug":"world-models-via-policy-guided-trajectory","title":"World Models via Policy-Guided Trajectory Diffusion","date":"2023-12-13","arxiv_id":"2312.08533","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":2,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 2 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/world-models-via-policy-guided-trajectory#ran","syntology_url":"https://syntology.ai/paper/2312.08533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08533"}},"official":{"repos":["marc-rigter/polygrad-world-models"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-perspective-of-q-value-estimation-on","slug":"a-perspective-of-q-value-estimation-on","title":"A Perspective of Q-value Estimation on Offline-to-Online Reinforcement Learning","date":"2023-12-12","arxiv_id":"2312.07685","repositories_listed":1,"syntology":null},{"url":"/paper/optimistic-multi-agent-policy-gradient-for","slug":"optimistic-multi-agent-policy-gradient-for","title":"Optimistic Multi-Agent Policy Gradient","date":"2023-11-03","arxiv_id":"2311.01953","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-multi-agent-policy-gradient-for#ran","syntology_url":"https://syntology.ai/paper/2311.01953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01953"}},"official":{"repos":["wenshuaizhao/optimappo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-models-are-zero-shot-reward","slug":"vision-language-models-are-zero-shot-reward","title":"Vision-Language Models are Zero-Shot Reward Models for Reinforcement Learning","date":"2023-10-19","arxiv_id":"2310.12921","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vision-language-models-are-zero-shot-reward#ran","syntology_url":"https://syntology.ai/paper/2310.12921","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12921"}},"official":{"repos":["alignmentresearch/vlmrm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lightzero-a-unified-benchmark-for-monte-carlo-1","slug":"lightzero-a-unified-benchmark-for-monte-carlo-1","title":"LightZero: A Unified Benchmark for Monte Carlo Tree Search in General Sequential Decision Scenarios","date":"2023-10-12","arxiv_id":"2310.08348","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-from-purified","slug":"imitation-learning-from-purified","title":"Imitation Learning from Purified Demonstrations","date":"2023-10-11","arxiv_id":"2310.07143","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-from-purified#ran","syntology_url":"https://syntology.ai/paper/2310.07143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07143"}},"official":{"repos":["yunke-wang/dp-il"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/practical-probabilistic-model-based-deep","slug":"practical-probabilistic-model-based-deep","title":"Practical Probabilistic Model-based Deep Reinforcement Learning by Integrating Dropout Uncertainty and Trajectory Sampling","date":"2023-09-20","arxiv_id":"2309.11089","repositories_listed":1,"syntology":null},{"url":"/paper/text2reward-automated-dense-reward-function","slug":"text2reward-automated-dense-reward-function","title":"Text2Reward: Reward Shaping with Language Models for Reinforcement Learning","date":"2023-09-20","arxiv_id":"2309.11489","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/text2reward-automated-dense-reward-function#ran","syntology_url":"https://syntology.ai/paper/2309.11489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11489"}},"official":{"repos":["xlang-ai/text2reward"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/heterogeneous-multi-agent-reinforcement-1","slug":"heterogeneous-multi-agent-reinforcement-1","title":"Heterogeneous Multi-Agent Reinforcement Learning via Mirror Descent Policy Optimization","date":"2023-08-13","arxiv_id":"2308.06741","repositories_listed":1,"syntology":null},{"url":"/paper/bierl-a-meta-evolutionary-reinforcement","slug":"bierl-a-meta-evolutionary-reinforcement","title":"BiERL: A Meta Evolutionary Reinforcement Learning Framework via Bilevel Optimization","date":"2023-08-01","arxiv_id":"2308.01207","repositories_listed":1,"syntology":null},{"url":"/paper/variance-control-for-distributional","slug":"variance-control-for-distributional","title":"Variance Control for Distributional Reinforcement Learning","date":"2023-07-30","arxiv_id":"2307.16152","repositories_listed":1,"syntology":null},{"url":"/paper/offline-multi-agent-reinforcement-learning-1","slug":"offline-multi-agent-reinforcement-learning-1","title":"Offline Multi-Agent Reinforcement Learning with Implicit Global-to-Local Value Regularization","date":"2023-07-21","arxiv_id":"2307.11620","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/offline-multi-agent-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2307.11620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11620"}},"official":{"repos":["zhengyinan-air/omiga"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-reinforcement-learning-techniques","slug":"exploring-reinforcement-learning-techniques","title":"Exploring reinforcement learning techniques for discrete and continuous control tasks in the MuJoCo environment","date":"2023-07-20","arxiv_id":"2307.11166","repositories_listed":1,"syntology":null},{"url":"/paper/natural-actor-critic-for-robust-reinforcement","slug":"natural-actor-critic-for-robust-reinforcement","title":"Natural Actor-Critic for Robust Reinforcement Learning with Function Approximation","date":"2023-07-17","arxiv_id":"2307.08875","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/natural-actor-critic-for-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2307.08875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08875"}},"official":{"repos":["tliu1997/rnac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-non-markovian-decision-making-from","slug":"learning-non-markovian-decision-making-from","title":"Learning non-Markovian Decision-Making from State-only Sequences","date":"2023-06-27","arxiv_id":"2306.15156","repositories_listed":1,"syntology":null},{"url":"/paper/adastop-sequential-testing-for-efficient-and","slug":"adastop-sequential-testing-for-efficient-and","title":"AdaStop: adaptive statistical testing for sound comparisons of Deep RL agents","date":"2023-06-19","arxiv_id":"2306.10882","repositories_listed":1,"syntology":null},{"url":"/paper/maximum-entropy-heterogeneous-agent-mirror","slug":"maximum-entropy-heterogeneous-agent-mirror","title":"Maximum Entropy Heterogeneous-Agent Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10715","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-on-policy-imitation-learning","slug":"sample-efficient-on-policy-imitation-learning","title":"Mimicking Better by Matching the Approximate Action Distribution","date":"2023-06-16","arxiv_id":"2306.09805","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-memory-decision-transformer","slug":"recurrent-memory-decision-transformer","title":"Recurrent Action Transformer with Memory","date":"2023-06-15","arxiv_id":"2306.09459","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-memory-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2306.09459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09459"}},"official":{"repos":["airi-institute/rate"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mildly-constrained-evaluation-policy-for","slug":"mildly-constrained-evaluation-policy-for","title":"Mildly Constrained Evaluation Policy for Offline Reinforcement Learning","date":"2023-06-06","arxiv_id":"2306.03680","repositories_listed":1,"syntology":null},{"url":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","repositories_listed":1,"syntology":null},{"url":"/paper/maximize-to-explore-one-objective-function","slug":"maximize-to-explore-one-objective-function","title":"Maximize to Explore: One Objective Function Fusing Estimation, Planning, and Exploration","date":"2023-05-29","arxiv_id":"2305.18258","repositories_listed":1,"syntology":null},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/off-policy-average-reward-actor-critic-with","slug":"off-policy-average-reward-actor-critic-with","title":"Off-Policy Average Reward Actor-Critic with Deterministic Policy Search","date":"2023-05-20","arxiv_id":"2305.12239","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/off-policy-average-reward-actor-critic-with#ran","syntology_url":"https://syntology.ai/paper/2305.12239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12239"}},"official":null}},{"url":"/paper/client-selection-for-federated-policy","slug":"client-selection-for-federated-policy","title":"Client Selection for Federated Policy Optimization with Environment Heterogeneity","date":"2023-05-18","arxiv_id":"2305.10978","repositories_listed":1,"syntology":null},{"url":"/paper/simple-noisy-environment-augmentation-for","slug":"simple-noisy-environment-augmentation-for","title":"Simple Noisy Environment Augmentation for Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.02882","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-pareto-efficient-decision-making-via","slug":"scaling-pareto-efficient-decision-making-via","title":"Scaling Pareto-Efficient Decision Making Via Offline Multi-Objective RL","date":"2023-04-30","arxiv_id":"2305.00567","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-pareto-efficient-decision-making-via#ran","syntology_url":"https://syntology.ai/paper/2305.00567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.00567"}},"official":{"repos":["baitingzbt/peda"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/feudal-graph-reinforcement-learning","slug":"feudal-graph-reinforcement-learning","title":"Feudal Graph Reinforcement Learning","date":"2023-04-11","arxiv_id":"2304.05099","repositories_listed":1,"syntology":null},{"url":"/paper/merging-decision-transformers-weight","slug":"merging-decision-transformers-weight","title":"Merging Decision Transformers: Weight Averaging for Forming Multi-Task Policies","date":"2023-03-14","arxiv_id":"2303.07551","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/merging-decision-transformers-weight#ran","syntology_url":"https://syntology.ai/paper/2303.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07551"}},"official":{"repos":["daniellawson9999/merging-decision-transformers"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/controlled-diversity-with-preference-towards","slug":"controlled-diversity-with-preference-towards","title":"Controlled Diversity with Preference : Towards Learning a Diverse Set of Desired Skills","date":"2023-03-07","arxiv_id":"2303.04592","repositories_listed":1,"syntology":null},{"url":"/paper/decision-transformer-under-random-frame","slug":"decision-transformer-under-random-frame","title":"Decision Transformer under Random Frame Dropping","date":"2023-03-03","arxiv_id":"2303.03391","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decision-transformer-under-random-frame#ran","syntology_url":"https://syntology.ai/paper/2303.03391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03391"}},"official":{"repos":["hukz18/defog"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/order-matters-agent-by-agent-policy","slug":"order-matters-agent-by-agent-policy","title":"Order Matters: Agent-by-agent Policy Optimization","date":"2023-02-13","arxiv_id":"2302.06205","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/order-matters-agent-by-agent-policy#ran","syntology_url":"https://syntology.ai/paper/2302.06205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06205"}},"official":{"repos":["xihuai18/A2PO-ICLR2023"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unlabeled-imperfect-demonstrations-in","slug":"unlabeled-imperfect-demonstrations-in","title":"Unlabeled Imperfect Demonstrations in Adversarial Imitation Learning","date":"2023-02-13","arxiv_id":"2302.06271","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-cooperative-multi-agent","slug":"attacking-cooperative-multi-agent","title":"Attacking Cooperative Multi-Agent Reinforcement Learning by Adversarial Minority Influence","date":"2023-02-07","arxiv_id":"2302.03322","repositories_listed":1,"syntology":null},{"url":"/paper/sample-dropout-a-simple-yet-effective","slug":"sample-dropout-a-simple-yet-effective","title":"Sample Dropout: A Simple yet Effective Variance Reduction Technique in Deep Policy Optimization","date":"2023-02-05","arxiv_id":"2302.02299","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/sample-dropout-a-simple-yet-effective#ran","syntology_url":"https://syntology.ai/paper/2302.02299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02299"}},"official":{"repos":["linzichuan/sdpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptdiffuser-diffusion-models-as-adaptive","slug":"adaptdiffuser-diffusion-models-as-adaptive","title":"AdaptDiffuser: Diffusion Models as Adaptive Self-evolving Planners","date":"2023-02-03","arxiv_id":"2302.01877","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptdiffuser-diffusion-models-as-adaptive#ran","syntology_url":"https://syntology.ai/paper/2302.01877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01877"}},"official":{"repos":["Liang-ZX/adaptdiffuser"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/joint-action-loss-for-proximal-policy","slug":"joint-action-loss-for-proximal-policy","title":"Joint action loss for proximal policy optimization","date":"2023-01-26","arxiv_id":"2301.10919","repositories_listed":1,"syntology":null},{"url":"/paper/partial-advantage-estimator-for-proximal","slug":"partial-advantage-estimator-for-proximal","title":"Partial advantage estimator for proximal policy optimization","date":"2023-01-26","arxiv_id":"2301.10920","repositories_listed":1,"syntology":null},{"url":"/paper/which-experiences-are-influential-for-your","slug":"which-experiences-are-influential-for-your","title":"Which Experiences Are Influential for Your Agent? Policy Iteration with Turn-over Dropout","date":"2023-01-26","arxiv_id":"2301.11168","repositories_listed":1,"syntology":null},{"url":"/paper/pontryagin-optimal-controller-via-neural","slug":"pontryagin-optimal-controller-via-neural","title":"Pontryagin Optimal Control via Neural Networks","date":"2022-12-30","arxiv_id":"2212.14566","repositories_listed":1,"syntology":null},{"url":"/paper/td3-with-reverse-kl-regularizer-for-offline","slug":"td3-with-reverse-kl-regularizer-for-offline","title":"TD3 with Reverse KL Regularizer for Offline Reinforcement Learning from Mixed Datasets","date":"2022-12-05","arxiv_id":"2212.02125","repositories_listed":1,"syntology":null},{"url":"/paper/out-of-dynamics-imitation-learning-from","slug":"out-of-dynamics-imitation-learning-from","title":"Out-of-Dynamics Imitation Learning from Multimodal Demonstrations","date":"2022-11-13","arxiv_id":"2211.06839","repositories_listed":1,"syntology":null},{"url":"/paper/robust-offline-reinforcement-learning-with","slug":"robust-offline-reinforcement-learning-with","title":"Robust Offline Reinforcement Learning with Gradient Penalty and Constraint Relaxation","date":"2022-10-19","arxiv_id":"2210.10469","repositories_listed":1,"syntology":null},{"url":"/paper/wild-scav-benchmarking-fps-gaming-ai-on","slug":"wild-scav-benchmarking-fps-gaming-ai-on","title":"WILD-SCAV: Benchmarking FPS Gaming AI on Unity3D-based Environments","date":"2022-10-14","arxiv_id":"2210.09026","repositories_listed":1,"syntology":null},{"url":"/paper/monte-carlo-tree-search-based-variable","slug":"monte-carlo-tree-search-based-variable","title":"Monte Carlo Tree Search based Variable Selection for High Dimensional Bayesian Optimization","date":"2022-10-04","arxiv_id":"2210.01628","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-reuse-bias-in-off-policy-reinforcement","slug":"on-the-reuse-bias-in-off-policy-reinforcement","title":"On the Reuse Bias in Off-Policy Reinforcement Learning","date":"2022-09-15","arxiv_id":"2209.07074","repositories_listed":1,"syntology":null},{"url":"/paper/cyclic-policy-distillation-sample-efficient","slug":"cyclic-policy-distillation-sample-efficient","title":"Cyclic Policy Distillation: Sample-Efficient Sim-to-Real Reinforcement Learning with Domain Randomization","date":"2022-07-29","arxiv_id":"2207.14561","repositories_listed":1,"syntology":null},{"url":"/paper/learning-bipedal-walking-on-planned-footsteps","slug":"learning-bipedal-walking-on-planned-footsteps","title":"Learning Bipedal Walking On Planned Footsteps For Humanoid Robots","date":"2022-07-26","arxiv_id":"2207.12644","repositories_listed":1,"syntology":null},{"url":"/paper/live-in-the-moment-learning-dynamics-model","slug":"live-in-the-moment-learning-dynamics-model","title":"Live in the Moment: Learning Dynamics Model Adapted to Evolving Policy","date":"2022-07-25","arxiv_id":"2207.12141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/live-in-the-moment-learning-dynamics-model#ran","syntology_url":"https://syntology.ai/paper/2207.12141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12141"}},"official":{"repos":["si0wang/pdml"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/short-term-plasticity-neurons-learning-to","slug":"short-term-plasticity-neurons-learning-to","title":"Short-Term Plasticity Neurons Learning to Learn and Forget","date":"2022-06-28","arxiv_id":"2206.14048","repositories_listed":1,"syntology":null},{"url":"/paper/cgar-critic-guided-action-redistribution-in","slug":"cgar-critic-guided-action-redistribution-in","title":"CGAR: Critic Guided Action Redistribution in Reinforcement Leaning","date":"2022-06-23","arxiv_id":"2206.11494","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-reinforcement-learning-via-2","slug":"towards-safe-reinforcement-learning-via-2","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value-at-Risk","date":"2022-06-09","arxiv_id":"2206.04436","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-reward-poisoning-attacks-on-online#ran","syntology_url":"https://syntology.ai/paper/2205.14842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14842"}},"official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-is-a","slug":"multi-agent-reinforcement-learning-is-a","title":"Multi-Agent Reinforcement Learning is a Sequence Modeling Problem","date":"2022-05-30","arxiv_id":"2205.14953","repositories_listed":1,"syntology":null},{"url":"/paper/tasil-taylor-series-imitation-learning","slug":"tasil-taylor-series-imitation-learning","title":"TaSIL: Taylor Series Imitation Learning","date":"2022-05-30","arxiv_id":"2205.14812","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tasil-taylor-series-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2205.14812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14812"}},"official":{"repos":["unstable-zeros/tasil"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/arlo-a-framework-for-automated-reinforcement","slug":"arlo-a-framework-for-automated-reinforcement","title":"ARLO: A Framework for Automated Reinforcement Learning","date":"2022-05-20","arxiv_id":"2205.10416","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-from-observations-under-1","slug":"imitation-learning-from-observations-under-1","title":"Imitation Learning from Observations under Transition Model Disparity","date":"2022-04-25","arxiv_id":"2204.11446","repositories_listed":1,"syntology":null},{"url":"/paper/jorldy-a-fully-customizable-open-source","slug":"jorldy-a-fully-customizable-open-source","title":"JORLDY: a fully customizable open source framework for reinforcement learning","date":"2022-04-11","arxiv_id":"2204.04892","repositories_listed":1,"syntology":null},{"url":"/paper/value-gradient-weighted-model-based-1","slug":"value-gradient-weighted-model-based-1","title":"Value Gradient weighted Model-Based Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01464","repositories_listed":1,"syntology":null},{"url":"/paper/deconstructing-the-inductive-biases-of-1","slug":"deconstructing-the-inductive-biases-of-1","title":"Deconstructing the Inductive Biases of Hamiltonian Neural Networks","date":"2022-02-10","arxiv_id":"2202.04836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deconstructing-the-inductive-biases-of-1#ran","syntology_url":"https://syntology.ai/paper/2202.04836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04836"}},"official":{"repos":["ngruver/decon-hnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lipschitz-constrained-unsupervised-skill-1","slug":"lipschitz-constrained-unsupervised-skill-1","title":"Lipschitz-constrained Unsupervised Skill Discovery","date":"2022-02-02","arxiv_id":"2202.00914","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lipschitz-constrained-unsupervised-skill-1#ran","syntology_url":"https://syntology.ai/paper/2202.00914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.00914"}},"official":null}},{"url":"/paper/dns-determinantal-point-process-based-neural","slug":"dns-determinantal-point-process-based-neural","title":"DNS: Determinantal Point Process Based Neural Network Sampler for Ensemble Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13357","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dns-determinantal-point-process-based-neural#ran","syntology_url":"https://syntology.ai/paper/2201.13357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13357"}},"official":{"repos":["IntelLabs/DNS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/comparing-model-free-and-model-based","slug":"comparing-model-free-and-model-based","title":"Comparing Model-free and Model-based Algorithms for Offline Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05433","repositories_listed":1,"syntology":null},{"url":"/paper/self-reward-design-with-fine-grained-1","slug":"self-reward-design-with-fine-grained-1","title":"Self Reward Design with Fine-grained Interpretability","date":"2021-12-30","arxiv_id":"2112.15034","repositories_listed":1,"syntology":null},{"url":"/paper/cem-gd-cross-entropy-method-with-gradient","slug":"cem-gd-cross-entropy-method-with-gradient","title":"CEM-GD: Cross-Entropy Method with Gradient Descent Planner for Model-Based Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07746","repositories_listed":1,"syntology":null},{"url":"/paper/ostrichrl-a-musculoskeletal-ostrich","slug":"ostrichrl-a-musculoskeletal-ostrich","title":"OstrichRL: A Musculoskeletal Ostrich Simulation to Study Bio-mechanical Locomotion","date":"2021-12-11","arxiv_id":"2112.06061","repositories_listed":1,"syntology":null},{"url":"/paper/residual-pathway-priors-for-soft-equivariance-1","slug":"residual-pathway-priors-for-soft-equivariance-1","title":"Residual Pathway Priors for Soft Equivariance Constraints","date":"2021-12-02","arxiv_id":"2112.01388","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/residual-pathway-priors-for-soft-equivariance-1#ran","syntology_url":"https://syntology.ai/paper/2112.01388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.01388"}},"official":{"repos":["mfinzi/residual-pathway-priors"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-domain-adaptation-for-cost","slug":"cross-modal-domain-adaptation-for-cost","title":"Cross-modal Domain Adaptation for Cost-Efficient Visual Reinforcement Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/edge-explaining-deep-reinforcement-learning","slug":"edge-explaining-deep-reinforcement-learning","title":"EDGE: Explaining Deep Reinforcement Learning Policies","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/offline-model-based-adaptable-policy-learning","slug":"offline-model-based-adaptable-policy-learning","title":"Offline Model-based Adaptable Policy Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continuous-control-with-ensemble-deep-1","slug":"continuous-control-with-ensemble-deep-1","title":"Continuous Control With Ensemble Deep Deterministic Policy Gradients","date":"2021-11-30","arxiv_id":"2111.15382","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-decision-transformer-for-offline","slug":"generalized-decision-transformer-for-offline","title":"Generalized Decision Transformer for Offline Hindsight Information Matching","date":"2021-11-19","arxiv_id":"2111.10364","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2111.10364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10364"}},"official":{"repos":["frt03/generalized_dt"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/robust-deep-reinforcement-learning-for-1","slug":"robust-deep-reinforcement-learning-for-1","title":"Robust Deep Reinforcement Learning for Quadcopter Control","date":"2021-11-06","arxiv_id":"2111.03915","repositories_listed":1,"syntology":null},{"url":"/paper/time-discretization-invariant-safe-action","slug":"time-discretization-invariant-safe-action","title":"Time Discretization-Invariant Safe Action Repetition for Policy Gradient Methods","date":"2021-11-06","arxiv_id":"2111.03941","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/time-discretization-invariant-safe-action#ran","syntology_url":"https://syntology.ai/paper/2111.03941","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.03941"}},"official":{"repos":["artberryx/SAR"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/conditioning-sparse-variational-gaussian","slug":"conditioning-sparse-variational-gaussian","title":"Conditioning Sparse Variational Gaussian Processes for Online Decision-making","date":"2021-10-28","arxiv_id":"2110.15172","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":3,"n_ran_checked":4,"n_instrument":7,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"11 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/conditioning-sparse-variational-gaussian#ran","syntology_url":"https://syntology.ai/paper/2110.15172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.15172"}},"official":{"repos":["wjmaddox/online_vargp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/balancing-value-underestimation-and","slug":"balancing-value-underestimation-and","title":"Balancing Value Underestimation and Overestimation with Realistic Actor-Critic","date":"2021-10-19","arxiv_id":"2110.09712","repositories_listed":1,"syntology":null},{"url":"/paper/parameter-free-deterministic-reduction-of-the","slug":"parameter-free-deterministic-reduction-of-the","title":"Parameter-free Reduction of the Estimation Bias in Deep Reinforcement Learning for Deterministic Policy Gradients","date":"2021-09-24","arxiv_id":"2109.11788","repositories_listed":1,"syntology":null},{"url":"/paper/settling-the-variance-of-multi-agent-policy","slug":"settling-the-variance-of-multi-agent-policy","title":"Settling the Variance of Multi-Agent Policy Gradients","date":"2021-08-19","arxiv_id":"2108.08612","repositories_listed":1,"syntology":null},{"url":"/paper/a-functional-mirror-ascent-view-of-policy","slug":"a-functional-mirror-ascent-view-of-policy","title":"A general class of surrogate functions for stable and efficient reinforcement learning","date":"2021-08-12","arxiv_id":"2108.05828","repositories_listed":1,"syntology":null}],"record_sha256":"8305c31358d99866521dbb1ed36b9034d162acb077dba898095c71af1fe58122","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}