{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/21","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":21,"pages_in_order":152,"rows_per_page":100,"rows":[2001,2100],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/20","next":"/task/reinforcement-learning-1/papers/22","papers":[{"url":"/paper/distance-weighted-supervised-learning-for","slug":"distance-weighted-supervised-learning-for","title":"Distance Weighted Supervised Learning for Offline Interaction Data","date":"2023-04-26","arxiv_id":"2304.13774","repositories_listed":1,"syntology":{"n":27,"n_ran":14,"n_constructed":0,"n_ran_checked":4,"n_instrument":10,"n_unverified":13,"n_honours":2,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 1 violated, 1 with no contract checked; 10 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/distance-weighted-supervised-learning-for#ran","syntology_url":"https://syntology.ai/paper/2304.13774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13774"}},"official":null}},{"url":"/paper/proximal-curriculum-for-reinforcement","slug":"proximal-curriculum-for-reinforcement","title":"Proximal Curriculum for Reinforcement Learning Agents","date":"2023-04-25","arxiv_id":"2304.12877","repositories_listed":1,"syntology":null},{"url":"/paper/deir-efficient-and-robust-exploration-through","slug":"deir-efficient-and-robust-exploration-through","title":"DEIR: Efficient and Robust Exploration through Discriminative-Model-Based Episodic Intrinsic Rewards","date":"2023-04-21","arxiv_id":"2304.10770","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-approaches-for-traffic","slug":"reinforcement-learning-approaches-for-traffic","title":"Reinforcement Learning Approaches for Traffic Signal Control under Missing Data","date":"2023-04-21","arxiv_id":"2304.10722","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-rl-theory-and-practice-with-the-1","slug":"bridging-rl-theory-and-practice-with-the-1","title":"Bridging RL Theory and Practice with the Effective Horizon","date":"2023-04-19","arxiv_id":"2304.09853","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-model-based-reinforcement","slug":"sample-efficient-model-based-reinforcement","title":"Sample-efficient Model-based Reinforcement Learning for Quantum Control","date":"2023-04-19","arxiv_id":"2304.09718","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-actor-critic-deep-reinforcement","slug":"benchmarking-actor-critic-deep-reinforcement","title":"Benchmarking Actor-Critic Deep Reinforcement Learning Algorithms for Robotics Control with Action Constraints","date":"2023-04-18","arxiv_id":"2304.08743","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-actor-critic-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2304.08743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08743"}},"official":{"repos":["omron-sinicx/action-constrained-rl-benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-offline-data-to-speed-up-reinforcement","slug":"using-offline-data-to-speed-up-reinforcement","title":"Using Offline Data to Speed Up Reinforcement Learning in Procedurally Generated Environments","date":"2023-04-18","arxiv_id":"2304.09825","repositories_listed":1,"syntology":null},{"url":"/paper/treec-a-method-to-generate-interpretable","slug":"treec-a-method-to-generate-interpretable","title":"TreeC: a method to generate interpretable energy management systems using a metaheuristic algorithm","date":"2023-04-17","arxiv_id":"2304.08310","repositories_listed":1,"syntology":null},{"url":"/paper/language-instructed-reinforcement-learning","slug":"language-instructed-reinforcement-learning","title":"Language Instructed Reinforcement Learning for Human-AI Coordination","date":"2023-04-13","arxiv_id":"2304.07297","repositories_listed":1,"syntology":null},{"url":"/paper/did-we-personalize-assessing-personalization","slug":"did-we-personalize-assessing-personalization","title":"Did we personalize? Assessing personalization by an online reinforcement learning algorithm using resampling","date":"2023-04-11","arxiv_id":"2304.05365","repositories_listed":1,"syntology":null},{"url":"/paper/feudal-graph-reinforcement-learning","slug":"feudal-graph-reinforcement-learning","title":"Feudal Graph Reinforcement Learning","date":"2023-04-11","arxiv_id":"2304.05099","repositories_listed":1,"syntology":null},{"url":"/paper/eagle-end-to-end-deep-reinforcement-learning","slug":"eagle-end-to-end-deep-reinforcement-learning","title":"Eagle: End-to-end Deep Reinforcement Learning based Autonomous Control of PTZ Cameras","date":"2023-04-10","arxiv_id":"2304.04356","repositories_listed":1,"syntology":null},{"url":"/paper/respect-reinforcement-learning-based-edge","slug":"respect-reinforcement-learning-based-edge","title":"RESPECT: Reinforcement Learning based Edge Scheduling on Pipelined Coral Edge TPUs","date":"2023-04-10","arxiv_id":"2304.04716","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-driven-trajectory-truncation-for","slug":"uncertainty-driven-trajectory-truncation-for","title":"Uncertainty-driven Trajectory Truncation for Data Augmentation in Offline Reinforcement Learning","date":"2023-04-10","arxiv_id":"2304.04660","repositories_listed":1,"syntology":null},{"url":"/paper/a-barrier-lyapunov-actor-critic-reinforcement","slug":"a-barrier-lyapunov-actor-critic-reinforcement","title":"Stable and Safe Reinforcement Learning via a Barrier-Lyapunov Actor-Critic Approach","date":"2023-04-08","arxiv_id":"2304.04066","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-bimanual-handover-and-rearrangement","slug":"efficient-bimanual-handover-and-rearrangement","title":"Efficient bimanual handover and rearrangement via symmetry-aware actor-critic learning","date":"2023-04-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/autorl-hyperparameter-landscapes","slug":"autorl-hyperparameter-landscapes","title":"AutoRL Hyperparameter Landscapes","date":"2023-04-05","arxiv_id":"2304.02396","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/autorl-hyperparameter-landscapes#ran","syntology_url":"https://syntology.ai/paper/2304.02396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.02396"}},"official":{"repos":["automl/autorl-landscape"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/managing-power-grids-through-topology-actions","slug":"managing-power-grids-through-topology-actions","title":"Managing power grids through topology actions: A comparative study between advanced rule-based and reinforcement learning agents","date":"2023-04-03","arxiv_id":"2304.00765","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-goal-reaching-reinforcement-learning","slug":"optimal-goal-reaching-reinforcement-learning","title":"Optimal Goal-Reaching Reinforcement Learning via Quasimetric Learning","date":"2023-04-03","arxiv_id":"2304.01203","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-goal-reaching-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.01203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01203"}},"official":{"repos":["quasimetric-learning/quasimetric-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-context-distribution-shift-in-task","slug":"on-context-distribution-shift-in-task","title":"On Context Distribution Shift in Task Representation Learning for Offline Meta RL","date":"2023-04-01","arxiv_id":"2304.00354","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-can-solve-computer-tasks","slug":"language-models-can-solve-computer-tasks","title":"Language Models can Solve Computer Tasks","date":"2023-03-30","arxiv_id":"2303.17491","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-can-solve-computer-tasks#ran","syntology_url":"https://syntology.ai/paper/2303.17491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17491"}},"official":{"repos":["posgnu/rci-agent"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pgx-hardware-accelerated-parallel-game-1","slug":"pgx-hardware-accelerated-parallel-game-1","title":"Pgx: Hardware-Accelerated Parallel Game Simulators for Reinforcement Learning","date":"2023-03-29","arxiv_id":"2303.17503","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/pgx-hardware-accelerated-parallel-game-1#ran","syntology_url":"https://syntology.ai/paper/2303.17503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17503"}},"official":{"repos":["sotetsuk/pgx"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/inverse-reinforcement-learning-without","slug":"inverse-reinforcement-learning-without","title":"Inverse Reinforcement Learning without Reinforcement Learning","date":"2023-03-26","arxiv_id":"2303.14623","repositories_listed":1,"syntology":null},{"url":"/paper/marl-jax-multi-agent-reinforcement-leaning","slug":"marl-jax-multi-agent-reinforcement-leaning","title":"marl-jax: Multi-Agent Reinforcement Leaning Framework","date":"2023-03-24","arxiv_id":"2303.13808","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-transport-for-offline-imitation","slug":"optimal-transport-for-offline-imitation","title":"Optimal Transport for Offline Imitation Learning","date":"2023-03-24","arxiv_id":"2303.13971","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-transport-for-offline-imitation#ran","syntology_url":"https://syntology.ai/paper/2303.13971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13971"}},"official":{"repos":["ethanluoyc/optimal_transport_reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-and-sample-efficient-reinforcement","slug":"safe-and-sample-efficient-reinforcement","title":"Safe and Sample-efficient Reinforcement Learning for Clustered Dynamic Environments","date":"2023-03-24","arxiv_id":"2303.14265","repositories_listed":1,"syntology":null},{"url":"/paper/data-might-be-enough-bridge-real-world","slug":"data-might-be-enough-bridge-real-world","title":"DataLight: Offline Data-Driven Traffic Signal Control","date":"2023-03-20","arxiv_id":"2303.10828","repositories_listed":1,"syntology":null},{"url":"/paper/imitating-graph-based-planning-with-goal","slug":"imitating-graph-based-planning-with-goal","title":"Imitating Graph-Based Planning with Goal-Conditioned Policies","date":"2023-03-20","arxiv_id":"2303.11166","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/imitating-graph-based-planning-with-goal#ran","syntology_url":"https://syntology.ai/paper/2303.11166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11166"}},"official":{"repos":["junsu-kim97/pig"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/clip4mc-an-rl-friendly-vision-language-model","slug":"clip4mc-an-rl-friendly-vision-language-model","title":"Reinforcement Learning Friendly Vision-Language Model for Minecraft","date":"2023-03-19","arxiv_id":"2303.10571","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip4mc-an-rl-friendly-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2303.10571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10571"}},"official":{"repos":["PKU-RL/CLIP4MC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-update-to-data-ratio-minimizing-world","slug":"dynamic-update-to-data-ratio-minimizing-world","title":"Dynamic Update-to-Data Ratio: Minimizing World Model Overfitting","date":"2023-03-17","arxiv_id":"2303.10144","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-update-to-data-ratio-minimizing-world#ran","syntology_url":"https://syntology.ai/paper/2303.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10144"}},"official":{"repos":["nicolinho/dutd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conditionally-optimistic-exploration-for","slug":"conditionally-optimistic-exploration-for","title":"Conditionally Optimistic Exploration for Cooperative Deep Multi-Agent Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09032","repositories_listed":1,"syntology":null},{"url":"/paper/act-then-measure-reinforcement-learning-for","slug":"act-then-measure-reinforcement-learning-for","title":"Act-Then-Measure: Reinforcement Learning for Partially Observable Environments with Active Measuring","date":"2023-03-14","arxiv_id":"2303.08271","repositories_listed":1,"syntology":null},{"url":"/paper/fast-rates-for-maximum-entropy-exploration","slug":"fast-rates-for-maximum-entropy-exploration","title":"Fast Rates for Maximum Entropy Exploration","date":"2023-03-14","arxiv_id":"2303.08059","repositories_listed":1,"syntology":null},{"url":"/paper/kernel-density-bayesian-inverse-reinforcement","slug":"kernel-density-bayesian-inverse-reinforcement","title":"Kernel Density Bayesian Inverse Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.06827","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kernel-density-bayesian-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.06827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06827"}},"official":{"repos":["bee-hive/kdbirl_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-world-models-are-happy-with","slug":"transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","repositories_listed":1,"syntology":{"n":25,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformer-based-world-models-are-happy-with#ran","syntology_url":"https://syntology.ai/paper/2303.07109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07109"}},"official":{"repos":["jrobine/twm"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/synthetic-experience-replay-1","slug":"synthetic-experience-replay-1","title":"Synthetic Experience Replay","date":"2023-03-12","arxiv_id":"2303.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/synthetic-experience-replay-1#ran","syntology_url":"https://syntology.ai/paper/2303.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06614"}},"official":{"repos":["conglu1997/SynthER"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/user-retention-oriented-recommendation-with","slug":"user-retention-oriented-recommendation-with","title":"User Retention-oriented Recommendation with Decision Transformer","date":"2023-03-11","arxiv_id":"2303.06347","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-foraging-strategies-can-be-learned","slug":"optimal-foraging-strategies-can-be-learned","title":"Optimal foraging strategies can be learned","date":"2023-03-10","arxiv_id":"2303.06050","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-populations-of-diverse-rl-agents","slug":"evolving-populations-of-diverse-rl-agents","title":"Evolving Populations of Diverse RL Agents with MAP-Elites","date":"2023-03-09","arxiv_id":"2303.12803","repositories_listed":1,"syntology":null},{"url":"/paper/mcts-geb-monte-carlo-tree-search-is-a-good-e","slug":"mcts-geb-monte-carlo-tree-search-is-a-good-e","title":"MCTS-GEB: Monte Carlo Tree Search is a Good E-graph Builder","date":"2023-03-08","arxiv_id":"2303.04651","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mcts-geb-monte-carlo-tree-search-is-a-good-e#ran","syntology_url":"https://syntology.ai/paper/2303.04651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.04651"}},"official":{"repos":["ucamrl/eqs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/raccer-towards-reachable-and-certain","slug":"raccer-towards-reachable-and-certain","title":"RACCER: Towards Reachable and Certain Counterfactual Explanations for Reinforcement Learning","date":"2023-03-08","arxiv_id":"2303.04475","repositories_listed":1,"syntology":null},{"url":"/paper/a-multiplicative-value-function-for-safe-and","slug":"a-multiplicative-value-function-for-safe-and","title":"A Multiplicative Value Function for Safe and Efficient Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.04118","repositories_listed":1,"syntology":null},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-bipedal-walking-for-humanoids-with","slug":"learning-bipedal-walking-for-humanoids-with","title":"Learning Bipedal Walking for Humanoids with Current Feedback","date":"2023-03-07","arxiv_id":"2303.03724","repositories_listed":1,"syntology":null},{"url":"/paper/learning-when-to-treat-business-processes","slug":"learning-when-to-treat-business-processes","title":"Learning When to Treat Business Processes: Prescriptive Process Monitoring with Causal Inference and Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03572","repositories_listed":1,"syntology":null},{"url":"/paper/zeroth-order-optimization-meets-human","slug":"zeroth-order-optimization-meets-human","title":"Zeroth-Order Optimization Meets Human Feedback: Provable Learning via Ranking Oracles","date":"2023-03-07","arxiv_id":"2303.03751","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/zeroth-order-optimization-meets-human#ran","syntology_url":"https://syntology.ai/paper/2303.03751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03751"}},"official":{"repos":["TZW1998/Taming-Stable-Diffusion-with-Human-Ranking-Feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-via-probabilistic","slug":"safe-reinforcement-learning-via-probabilistic","title":"Safe Reinforcement Learning via Probabilistic Logic Shields","date":"2023-03-06","arxiv_id":"2303.03226","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safe-reinforcement-learning-via-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2303.03226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03226"}},"official":null}},{"url":"/paper/bounding-the-optimal-value-function-in","slug":"bounding-the-optimal-value-function-in","title":"Bounding the Optimal Value Function in Compositional Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02557","repositories_listed":1,"syntology":null},{"url":"/paper/improved-sample-complexity-bounds-for","slug":"improved-sample-complexity-bounds-for","title":"Improved Sample Complexity Bounds for Distributionally Robust Reinforcement Learning","date":"2023-03-05","arxiv_id":"2303.02783","repositories_listed":1,"syntology":null},{"url":"/paper/swim-a-general-purpose-high-performing-and","slug":"swim-a-general-purpose-high-performing-and","title":"Swim: A General-Purpose, High-Performing, and Efficient Activation Function for Locomotion Control Tasks","date":"2023-03-05","arxiv_id":"2303.02640","repositories_listed":1,"syntology":null},{"url":"/paper/neural-airport-ground-handling","slug":"neural-airport-ground-handling","title":"Neural Airport Ground Handling","date":"2023-03-04","arxiv_id":"2303.02442","repositories_listed":1,"syntology":{"n":10,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/neural-airport-ground-handling#ran","syntology_url":"https://syntology.ai/paper/2303.02442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02442"}},"official":{"repos":["royalskye/agh"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/corl-environment-creation-and-management","slug":"corl-environment-creation-and-management","title":"CoRL: Environment Creation and Management Focused on System Integration","date":"2023-03-03","arxiv_id":"2303.02182","repositories_listed":1,"syntology":null},{"url":"/paper/ghq-grouped-hybrid-q-learning-for","slug":"ghq-grouped-hybrid-q-learning-for","title":"GHQ: Grouped Hybrid Q Learning for Heterogeneous Cooperative Multi-agent Reinforcement Learning","date":"2023-03-02","arxiv_id":"2303.01070","repositories_listed":1,"syntology":null},{"url":"/paper/preference-transformer-modeling-human","slug":"preference-transformer-modeling-human","title":"Preference Transformer: Modeling Human Preferences using Transformers for RL","date":"2023-03-02","arxiv_id":"2303.00957","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/preference-transformer-modeling-human#ran","syntology_url":"https://syntology.ai/paper/2303.00957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.00957"}},"official":{"repos":["csmile-1006/preferencetransformer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-guided-multi-objective","slug":"reinforcement-learning-guided-multi-objective","title":"Reinforcement Learning Guided Multi-Objective Exam Paper Generation","date":"2023-03-02","arxiv_id":"2303.01042","repositories_listed":1,"syntology":null},{"url":"/paper/ls-iq-implicit-reward-regularization-for","slug":"ls-iq-implicit-reward-regularization-for","title":"LS-IQ: Implicit Reward Regularization for Inverse Reinforcement Learning","date":"2023-03-01","arxiv_id":"2303.00599","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-multiple-abstractions-in-episodic","slug":"exploiting-multiple-abstractions-in-episodic","title":"Exploiting Multiple Abstractions in Episodic RL via Reward Shaping","date":"2023-02-28","arxiv_id":"2303.00516","repositories_listed":1,"syntology":null},{"url":"/paper/human-inspired-framework-to-accelerate","slug":"human-inspired-framework-to-accelerate","title":"Human-Inspired Framework to Accelerate Reinforcement Learning","date":"2023-02-28","arxiv_id":"2303.08115","repositories_listed":1,"syntology":null},{"url":"/paper/reward-design-with-language-models","slug":"reward-design-with-language-models","title":"Reward Design with Language Models","date":"2023-02-27","arxiv_id":"2303.00001","repositories_listed":1,"syntology":null},{"url":"/paper/systematic-rectification-of-language-models","slug":"systematic-rectification-of-language-models","title":"Systematic Rectification of Language Models via Dead-end Analysis","date":"2023-02-27","arxiv_id":"2302.14003","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/systematic-rectification-of-language-models#ran","syntology_url":"https://syntology.ai/paper/2302.14003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14003"}},"official":{"repos":["mcao516/rectification-lm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evotorch-scalable-evolutionary-computation-in","slug":"evotorch-scalable-evolutionary-computation-in","title":"EvoTorch: Scalable Evolutionary Computation in Python","date":"2023-02-24","arxiv_id":"2302.12600","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/evotorch-scalable-evolutionary-computation-in#ran","syntology_url":"https://syntology.ai/paper/2302.12600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12600"}},"official":{"repos":["nnaisense/evotorch"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-efficient-neural-offline","slug":"provably-efficient-neural-offline","title":"VIPeR: Provably Efficient Algorithm for Offline RL with Neural Function Approximation","date":"2023-02-24","arxiv_id":"2302.12780","repositories_listed":1,"syntology":null},{"url":"/paper/diverse-policy-optimization-for-structured","slug":"diverse-policy-optimization-for-structured","title":"Diverse Policy Optimization for Structured Action Space","date":"2023-02-23","arxiv_id":"2302.11917","repositories_listed":1,"syntology":null},{"url":"/paper/energy-harvesting-reconfigurable-intelligent","slug":"energy-harvesting-reconfigurable-intelligent","title":"Energy Harvesting Reconfigurable Intelligent Surface for UAV Based on Robust Deep Reinforcement Learning","date":"2023-02-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/trustworthy-reinforcement-learning-for","slug":"trustworthy-reinforcement-learning-for","title":"Constrained Reinforcement Learning using Distributional Representation for Trustworthy Quadrotor UAV Tracking Control","date":"2023-02-22","arxiv_id":"2302.11694","repositories_listed":1,"syntology":null},{"url":"/paper/assessment-of-reinforcement-learning-for","slug":"assessment-of-reinforcement-learning-for","title":"Assessment of Reinforcement Learning for Macro Placement","date":"2023-02-21","arxiv_id":"2302.11014","repositories_listed":1,"syntology":null},{"url":"/paper/mac-po-multi-agent-experience-replay-via","slug":"mac-po-multi-agent-experience-replay-via","title":"MAC-PO: Multi-Agent Experience Replay via Collective Priority Optimization","date":"2023-02-21","arxiv_id":"2302.10418","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-bayes-reinforcement-learning","slug":"minimax-bayes-reinforcement-learning","title":"Minimax-Bayes Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10831","repositories_listed":1,"syntology":null},{"url":"/paper/potential-based-reward-shaping-for-learning","slug":"potential-based-reward-shaping-for-learning","title":"Learning to Play Text-based Adventure Games with Maximum Entropy Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10720","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-cost","slug":"deep-reinforcement-learning-for-cost","title":"Deep Reinforcement Learning for Cost-Effective Medical Diagnosis","date":"2023-02-20","arxiv_id":"2302.10261","repositories_listed":1,"syntology":null},{"url":"/paper/multiagent-inverse-reinforcement-learning-via","slug":"multiagent-inverse-reinforcement-learning-via","title":"Multiagent Inverse Reinforcement Learning via Theory of Mind Reasoning","date":"2023-02-20","arxiv_id":"2302.10238","repositories_listed":1,"syntology":null},{"url":"/paper/take-me-home-reversing-distribution-shifts","slug":"take-me-home-reversing-distribution-shifts","title":"DC4L: Distribution Shift Recovery via Data-Driven Control for Deep Learning Models","date":"2023-02-20","arxiv_id":"2302.10341","repositories_listed":1,"syntology":null},{"url":"/paper/auto-gov-learning-based-on-chain-governance","slug":"auto-gov-learning-based-on-chain-governance","title":"Auto.gov: Learning-based Governance for Decentralized Finance (DeFi)","date":"2023-02-19","arxiv_id":"2302.09551","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-in-visual-reinforcement","slug":"generalization-in-visual-reinforcement","title":"Generalization in Visual Reinforcement Learning with the Reward Sequence Distribution","date":"2023-02-19","arxiv_id":"2302.09601","repositories_listed":1,"syntology":null},{"url":"/paper/post-episodic-reinforcement-learning","slug":"post-episodic-reinforcement-learning","title":"Post Reinforcement Learning Inference","date":"2023-02-17","arxiv_id":"2302.08854","repositories_listed":1,"syntology":null},{"url":"/paper/swapped-goal-conditioned-offline","slug":"swapped-goal-conditioned-offline","title":"Swapped goal-conditioned offline reinforcement learning","date":"2023-02-17","arxiv_id":"2302.08865","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-from-arbitrary-experience-a-dual","slug":"imitation-from-arbitrary-experience-a-dual","title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","date":"2023-02-16","arxiv_id":"2302.08560","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-from-arbitrary-experience-a-dual#ran","syntology_url":"https://syntology.ai/paper/2302.08560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08560"}},"official":{"repos":["hari-sikchi/DVL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tuning-computer-vision-models-with-task","slug":"tuning-computer-vision-models-with-task","title":"Tuning computer vision models with task rewards","date":"2023-02-16","arxiv_id":"2302.08242","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-decision-transformer-for-offline","slug":"constrained-decision-transformer-for-offline","title":"Constrained Decision Transformer for Offline Safe Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07351","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constrained-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2302.07351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07351"}},"official":{"repos":["liuzuxin/osrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-a-model-is-paramount-for-sample","slug":"learning-a-model-is-paramount-for-sample","title":"Learning a model is paramount for sample efficiency in reinforcement learning control of PDEs","date":"2023-02-14","arxiv_id":"2302.07160","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-through-2","slug":"robust-deep-reinforcement-learning-through-2","title":"Regret-Based Defense in Adversarial Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06912","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-2#ran","syntology_url":"https://syntology.ai/paper/2302.06912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06912"}},"official":{"repos":["romanbelaire/robust-ccer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semiconductor-fab-scheduling-with-self","slug":"semiconductor-fab-scheduling-with-self","title":"Semiconductor Fab Scheduling with Self-Supervised and Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07162","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-noise-filtering-with-dynamic-sparse#ran","syntology_url":"https://syntology.ai/paper/2302.06548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06548"}},"official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-pretraining-in-reinforcement-learning","slug":"guiding-pretraining-in-reinforcement-learning","title":"Guiding Pretraining in Reinforcement Learning with Large Language Models","date":"2023-02-13","arxiv_id":"2302.06692","repositories_listed":1,"syntology":null},{"url":"/paper/robust-representation-learning-by-clustering","slug":"robust-representation-learning-by-clustering","title":"Robust Representation Learning by Clustering with Bisimulation Metrics for Visual Reinforcement Learning with Distractions","date":"2023-02-12","arxiv_id":"2302.12003","repositories_listed":1,"syntology":null},{"url":"/paper/a-large-parametrized-space-of-meta","slug":"a-large-parametrized-space-of-meta","title":"Procedural generation of meta-reinforcement learning tasks","date":"2023-02-11","arxiv_id":"2302.05583","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-random-pre-training-with","slug":"cross-domain-random-pre-training-with","title":"Cross-domain Random Pre-training with Prototypes for Reinforcement Learning","date":"2023-02-11","arxiv_id":"2302.05614","repositories_listed":1,"syntology":null},{"url":"/paper/a-swat-based-reinforcement-learning-framework","slug":"a-swat-based-reinforcement-learning-framework","title":"A SWAT-based Reinforcement Learning Framework for Crop Management","date":"2023-02-10","arxiv_id":"2302.04988","repositories_listed":1,"syntology":null},{"url":"/paper/on-penalty-based-bilevel-gradient-descent","slug":"on-penalty-based-bilevel-gradient-descent","title":"On Penalty-based Bilevel Gradient Descent Method","date":"2023-02-10","arxiv_id":"2302.05185","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-penalty-based-bilevel-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/2302.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.05185"}},"official":{"repos":["hanshen95/penalized-bilevel-gradient-descent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-from-multiple-sensors","slug":"reinforcement-learning-from-multiple-sensors","title":"Combining Reconstruction and Contrastive Methods for Multimodal Representations in RL","date":"2023-02-10","arxiv_id":"2302.05342","repositories_listed":1,"syntology":null},{"url":"/paper/the-wisdom-of-hindsight-makes-language-models","slug":"the-wisdom-of-hindsight-makes-language-models","title":"The Wisdom of Hindsight Makes Language Models Better Instruction Followers","date":"2023-02-10","arxiv_id":"2302.05206","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-generative-adversarial-imitation","slug":"hierarchical-generative-adversarial-imitation","title":"Hierarchical Generative Adversarial Imitation Learning with Mid-level Input Generation for Autonomous Driving on Urban Environments","date":"2023-02-09","arxiv_id":"2302.04823","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-generative-adversarial-imitation#ran","syntology_url":"https://syntology.ai/paper/2302.04823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.04823"}},"official":{"repos":["gustavokcouto/hgail"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-complex-teamwork-tasks-using-a-sub","slug":"learning-complex-teamwork-tasks-using-a-sub","title":"Learning Complex Teamwork Tasks Using a Given Sub-task Decomposition","date":"2023-02-09","arxiv_id":"2302.04944","repositories_listed":1,"syntology":null},{"url":"/paper/maniskill2-a-unified-benchmark-for","slug":"maniskill2-a-unified-benchmark-for","title":"ManiSkill2: A Unified Benchmark for Generalizable Manipulation Skills","date":"2023-02-09","arxiv_id":"2302.04659","repositories_listed":1,"syntology":null},{"url":"/paper/raynet-a-simulation-platform-for-developing","slug":"raynet-a-simulation-platform-for-developing","title":"RayNet: A Simulation Platform for Developing Reinforcement Learning-Driven Network Protocols","date":"2023-02-09","arxiv_id":"2302.04519","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-goal-based-exploration-via-pruning","slug":"scaling-goal-based-exploration-via-pruning","title":"Scaling Goal-based Exploration via Pruning Proto-goals","date":"2023-02-09","arxiv_id":"2302.04693","repositories_listed":1,"syntology":null},{"url":"/paper/learning-graph-enhanced-commander-executor","slug":"learning-graph-enhanced-commander-executor","title":"Learning Graph-Enhanced Commander-Executor for Multi-Agent Navigation","date":"2023-02-08","arxiv_id":"2302.04094","repositories_listed":1,"syntology":null}],"record_sha256":"82b6538928fb54d3d65181f13718c21af2e0bbf7c3adacf7e4fde31c2c038c76","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}