{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/21","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":21,"pages_in_order":132,"rows_per_page":100,"rows":[2001,2100],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/20","next":"/task/reinforcement-learning/papers/22","papers":[{"url":"/paper/learning-to-optimize-for-reinforcement","slug":"learning-to-optimize-for-reinforcement","title":"Learning to Optimize for Reinforcement Learning","date":"2023-02-03","arxiv_id":"2302.01470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-optimize-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.01470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01470"}},"official":{"repos":["sail-sg/optim4rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-expansion-for-bridging-offline-to","slug":"policy-expansion-for-bridging-offline-to","title":"Policy Expansion for Bridging Offline-to-Online Reinforcement Learning","date":"2023-02-02","arxiv_id":"2302.00935","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-expansion-for-bridging-offline-to#ran","syntology_url":"https://syntology.ai/paper/2302.00935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00935"}},"official":{"repos":["haichao-zhang/pex"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/internally-rewarded-reinforcement-learning","slug":"internally-rewarded-reinforcement-learning","title":"Internally Rewarded Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00270","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internally-rewarded-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00270"}},"official":{"repos":["mengdi-li/internally-rewarded-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-3","slug":"a-reinforcement-learning-framework-for-3","title":"A Reinforcement Learning Framework for Dynamic Mediation Analysis","date":"2023-01-31","arxiv_id":"2301.13348","repositories_listed":1,"syntology":null},{"url":"/paper/execution-based-code-generation-using-deep","slug":"execution-based-code-generation-using-deep","title":"Execution-based Code Generation using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13816","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/execution-based-code-generation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2301.13816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13816"}},"official":{"repos":["reddy-lab-code-research/PPOCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-transport-perturbations-for-safe","slug":"optimal-transport-perturbations-for-safe","title":"Optimal Transport Perturbations for Safe Reinforcement Learning with Robustness Guarantees","date":"2023-01-31","arxiv_id":"2301.13375","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-online-reinforcement-learning-with","slug":"guiding-online-reinforcement-learning-with","title":"Guiding Online Reinforcement Learning with Action-Free Offline Pretraining","date":"2023-01-30","arxiv_id":"2301.12876","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2301.12876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12876"}},"official":{"repos":["vision-cair/af-guide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/importance-weighted-actor-critic-for-optimal-1","slug":"importance-weighted-actor-critic-for-optimal-1","title":"Importance Weighted Actor-Critic for Optimal Conservative Offline Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.12714","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/importance-weighted-actor-critic-for-optimal-1#ran","syntology_url":"https://syntology.ai/paper/2301.12714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12714"}},"official":{"repos":["zhuhl98/acrab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-multiple-epidemic-interventions-with","slug":"planning-multiple-epidemic-interventions-with","title":"Planning Multiple Epidemic Interventions with Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.12802","repositories_listed":1,"syntology":null},{"url":"/paper/apac-authorized-probability-controlled-actor","slug":"apac-authorized-probability-controlled-actor","title":"Constrained Policy Optimization with Explicit Behavior Density for Offline Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12130","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/apac-authorized-probability-controlled-actor#ran","syntology_url":"https://syntology.ai/paper/2301.12130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12130"}},"official":{"repos":["evalarzj/cped"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-intrinsic-reward-shaping-for","slug":"automatic-intrinsic-reward-shaping-for","title":"Automatic Intrinsic Reward Shaping for Exploration in Deep Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.10886","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-trust-region-based-safe","slug":"efficient-trust-region-based-safe","title":"Trust Region-Based Safe Distributional Reinforcement Learning for Multiple Constraints","date":"2023-01-26","arxiv_id":"2301.10923","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-trust-region-based-safe#ran","syntology_url":"https://syntology.ai/paper/2301.10923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.10923"}},"official":{"repos":["rllab-snu/safe-distributional-actor-critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-multiple-independent-advisors","slug":"learning-from-multiple-independent-advisors","title":"Learning from Multiple Independent Advisors in Multi-agent Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11153","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-aware-eligibility-traces-for-off","slug":"trajectory-aware-eligibility-traces-for-off","title":"Trajectory-Aware Eligibility Traces for Off-Policy Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11321","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trajectory-aware-eligibility-traces-for-off#ran","syntology_url":"https://syntology.ai/paper/2301.11321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11321"}},"official":{"repos":["brett-daley/trajectory-aware-etraces"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distributed-control-of-partial-differential","slug":"distributed-control-of-partial-differential","title":"Distributed Control of Partial Differential Equations Using Convolutional Reinforcement Learning","date":"2023-01-25","arxiv_id":"2301.10737","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-reinforcement-learning-for-2","slug":"constrained-reinforcement-learning-for-2","title":"Constrained Reinforcement Learning for Dexterous Manipulation","date":"2023-01-24","arxiv_id":"2301.09766","repositories_listed":1,"syntology":null},{"url":"/paper/the-configurable-tree-graph-ct-graph","slug":"the-configurable-tree-graph-ct-graph","title":"The configurable tree graph (CT-graph): measurable problems in partially observable and distal reward environments for lifelong reinforcement learning","date":"2023-01-21","arxiv_id":"2302.10887","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-path","slug":"deep-reinforcement-learning-based-path","title":"Deep-Reinforcement-Learning-based Path Planning for Industrial Robots using Distance Sensors as Observation","date":"2023-01-14","arxiv_id":"2301.05980","repositories_listed":1,"syntology":null},{"url":"/paper/mutation-testing-of-deep-reinforcement","slug":"mutation-testing-of-deep-reinforcement","title":"Mutation Testing of Deep Reinforcement Learning Based on Real Faults","date":"2023-01-13","arxiv_id":"2301.05651","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-online-multi-task-reinforcement","slug":"adversarial-online-multi-task-reinforcement","title":"Adversarial Online Multi-Task Reinforcement Learning","date":"2023-01-11","arxiv_id":"2301.04268","repositories_listed":1,"syntology":null},{"url":"/paper/hint-assisted-reinforcement-learning-an","slug":"hint-assisted-reinforcement-learning-an","title":"Hint assisted reinforcement learning: an application in radio astronomy","date":"2023-01-10","arxiv_id":"2301.03933","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-perceive-in-deep-model-free","slug":"learning-to-perceive-in-deep-model-free","title":"Learning to Perceive in Deep Model-Free Reinforcement Learning","date":"2023-01-10","arxiv_id":"2301.03730","repositories_listed":1,"syntology":null},{"url":"/paper/orbit-a-unified-simulation-framework-for","slug":"orbit-a-unified-simulation-framework-for","title":"Orbit: A Unified Simulation Framework for Interactive Robot Learning Environments","date":"2023-01-10","arxiv_id":"2301.04195","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/orbit-a-unified-simulation-framework-for#ran","syntology_url":"https://syntology.ai/paper/2301.04195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.04195"}},"official":{"repos":["NVIDIA-Omniverse/Orbit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-collective-intelligence-from-massive","slug":"emergent-collective-intelligence-from-massive","title":"Emergent collective intelligence from massive-agent cooperation and competition","date":"2023-01-04","arxiv_id":"2301.01609","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergent-collective-intelligence-from-massive#ran","syntology_url":"https://syntology.ai/paper/2301.01609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.01609"}},"official":{"repos":["hanmochen/lux-open"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robofriend-an-adpative-storytelling-robotic","slug":"robofriend-an-adpative-storytelling-robotic","title":"Robofriend: An Adpative Storytelling Robotic Teddy Bear - Technical Report","date":"2023-01-04","arxiv_id":"2301.01576","repositories_listed":1,"syntology":null},{"url":"/paper/environment-agnostic-representation-for","slug":"environment-agnostic-representation-for","title":"Environment Agnostic Representation for Visual Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fusing-pre-trained-language-models-with","slug":"fusing-pre-trained-language-models-with","title":"Fusing Pre-Trained Language Models With Multimodal Prompts Through Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/goal-guided-transformer-enabled-reinforcement","slug":"goal-guided-transformer-enabled-reinforcement","title":"Goal-Guided Transformer-Enabled Reinforcement Learning for Efficient Autonomous Navigation","date":"2023-01-01","arxiv_id":"2301.00362","repositories_listed":1,"syntology":null},{"url":"/paper/self-activating-neural-ensembles-for-1","slug":"self-activating-neural-ensembles-for-1","title":"Self-Activating Neural Ensembles for Continual Reinforcement Learning","date":"2022-12-31","arxiv_id":"2301.00141","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-success-induced","slug":"reinforcement-learning-with-success-induced","title":"Reinforcement Learning with Success Induced Task Prioritization","date":"2022-12-30","arxiv_id":"2301.00691","repositories_listed":1,"syntology":null},{"url":"/paper/risk-sensitive-policy-with-distributional","slug":"risk-sensitive-policy-with-distributional","title":"Risk-Sensitive Policy with Distributional Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14743","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-in-transformer-as-backbone-for","slug":"transformer-in-transformer-as-backbone-for","title":"Transformer in Transformer as Backbone for Deep Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14538","repositories_listed":1,"syntology":null},{"url":"/paper/lexicographic-multi-objective-reinforcement","slug":"lexicographic-multi-objective-reinforcement","title":"Lexicographic Multi-Objective Reinforcement Learning","date":"2022-12-28","arxiv_id":"2212.13769","repositories_listed":1,"syntology":null},{"url":"/paper/on-pathologies-in-kl-regularized-1","slug":"on-pathologies-in-kl-regularized-1","title":"On Pathologies in KL-Regularized Reinforcement Learning from Expert Demonstrations","date":"2022-12-28","arxiv_id":"2212.13936","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-pathologies-in-kl-regularized-1#ran","syntology_url":"https://syntology.ai/paper/2212.13936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13936"}},"official":{"repos":["conglu1997/nppac"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/strangeness-driven-exploration-in-multi-agent","slug":"strangeness-driven-exploration-in-multi-agent","title":"Strangeness-driven Exploration in Multi-Agent Reinforcement Learning","date":"2022-12-27","arxiv_id":"2212.13448","repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-representations-for-1","slug":"learning-generalizable-representations-for-1","title":"Learning Generalizable Representations for Reinforcement Learning via Adaptive Meta-learner of Behavioral Similarities","date":"2022-12-26","arxiv_id":"2212.13088","repositories_listed":1,"syntology":null},{"url":"/paper/example-guided-learning-of-stochastic-human","slug":"example-guided-learning-of-stochastic-human","title":"Example-guided learning of stochastic human driving policies using deep reinforcement learning","date":"2022-12-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/nars-vs-reinforcement-learning-ona-vs-q","slug":"nars-vs-reinforcement-learning-ona-vs-q","title":"NARS vs. Reinforcement learning: ONA vs. Q-Learning","date":"2022-12-23","arxiv_id":"2212.12517","repositories_listed":1,"syntology":null},{"url":"/paper/certified-policy-smoothing-for-cooperative","slug":"certified-policy-smoothing-for-cooperative","title":"Certified Policy Smoothing for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-22","arxiv_id":"2212.11746","repositories_listed":1,"syntology":null},{"url":"/paper/critic-guided-decoding-for-controlled-text","slug":"critic-guided-decoding-for-controlled-text","title":"Critic-Guided Decoding for Controlled Text Generation","date":"2022-12-21","arxiv_id":"2212.10938","repositories_listed":1,"syntology":null},{"url":"/paper/on-reinforcement-learning-for-the-game-of","slug":"on-reinforcement-learning-for-the-game-of","title":"On Reinforcement Learning for the Game of 2048","date":"2022-12-21","arxiv_id":"2212.11087","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-of-model-free-and-model-based","slug":"comparison-of-model-free-and-model-based","title":"Comparison of Model-Free and Model-Based Learning-Informed Planning for PointGoal Navigation","date":"2022-12-17","arxiv_id":"2212.08801","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-visual","slug":"offline-reinforcement-learning-for-visual","title":"Offline Reinforcement Learning for Visual Navigation","date":"2022-12-16","arxiv_id":"2212.08244","repositories_listed":1,"syntology":null},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/smacv2-an-improved-benchmark-for-cooperative-1","slug":"smacv2-an-improved-benchmark-for-cooperative-1","title":"SMACv2: An Improved Benchmark for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07489","repositories_listed":1,"syntology":null},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/effects-of-spectral-normalization-in-multi","slug":"effects-of-spectral-normalization-in-multi","title":"Effects of Spectral Normalization in Multi-agent Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effects-of-spectral-normalization-in-multi#ran","syntology_url":"https://syntology.ai/paper/2212.05331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05331"}},"official":{"repos":["kinalmehta/epymarl_spectral"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/specifying-behavior-preference-with-tiered","slug":"specifying-behavior-preference-with-tiered","title":"Tiered Reward: Designing Rewards for Specification and Fast Learning of Desired Behavior","date":"2022-12-07","arxiv_id":"2212.03733","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-risk-aware-bidding-with-budget","slug":"adaptive-risk-aware-bidding-with-budget","title":"Adaptive Risk-Aware Bidding with Budget Constraint in Display Advertising","date":"2022-12-06","arxiv_id":"2212.12533","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-side-chain-packing-arrangement-of","slug":"solving-the-side-chain-packing-arrangement-of","title":"Reinforcement Learning for Molecular Dynamics Optimization: A Stochastic Pontryagin Maximum Principle Approach","date":"2022-12-06","arxiv_id":"2212.03320","repositories_listed":1,"syntology":null},{"url":"/paper/state-space-closure-revisiting-endless-online","slug":"state-space-closure-revisiting-endless-online","title":"State Space Closure: Revisiting Endless Online Level Generation via Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.02951","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/physics-informed-model-based-reinforcement","slug":"physics-informed-model-based-reinforcement","title":"Physics-Informed Model-Based Reinforcement Learning","date":"2022-12-05","arxiv_id":"2212.02179","repositories_listed":1,"syntology":null},{"url":"/paper/stl-based-synthesis-of-feedback-controllers","slug":"stl-based-synthesis-of-feedback-controllers","title":"STL-Based Synthesis of Feedback Controllers Using Reinforcement Learning","date":"2022-12-02","arxiv_id":"2212.01022","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-through","slug":"efficient-reinforcement-learning-through","title":"Efficient Reinforcement Learning Through Trajectory Generation","date":"2022-11-30","arxiv_id":"2211.17249","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-bidding-strategy-in-display","slug":"real-time-bidding-strategy-in-display","title":"Real-time Bidding Strategy in Display Advertising: An Empirical Analysis","date":"2022-11-30","arxiv_id":"2212.02222","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-language-modeling-for-end-to-end","slug":"reinforced-language-modeling-for-end-to-end","title":"KRLS: Improving End-to-End Response Generation in Task Oriented Dialog with Reinforced Keywords Learning","date":"2022-11-30","arxiv_id":"2211.16773","repositories_listed":1,"syntology":null},{"url":"/paper/welfare-and-fairness-in-multi-objective","slug":"welfare-and-fairness-in-multi-objective","title":"Welfare and Fairness in Multi-objective Reinforcement Learning","date":"2022-11-30","arxiv_id":"2212.01382","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/welfare-and-fairness-in-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2212.01382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01382"}},"official":{"repos":["MuhangTian/Fair-MORL-AAMAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-estimation-from-multi-source-data","slug":"behavior-estimation-from-multi-source-data","title":"Behavior Estimation from Multi-Source Data for Offline Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.16078","repositories_listed":1,"syntology":null},{"url":"/paper/interpreting-primal-dual-algorithms-for","slug":"interpreting-primal-dual-algorithms-for","title":"Interpreting Primal-Dual Algorithms for Constrained Multiagent Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.16069","repositories_listed":1,"syntology":null},{"url":"/paper/quantile-constrained-reinforcement-learning-a","slug":"quantile-constrained-reinforcement-learning-a","title":"Quantile Constrained Reinforcement Learning: A Reinforcement Learning Framework Constraining Outage Probability","date":"2022-11-28","arxiv_id":"2211.15034","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quantile-constrained-reinforcement-learning-a#ran","syntology_url":"https://syntology.ai/paper/2211.15034","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15034"}},"official":{"repos":["wyjung0625/qcpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/applying-deep-reinforcement-learning-to-the","slug":"applying-deep-reinforcement-learning-to-the","title":"Applying Deep Reinforcement Learning to the HP Model for Protein Structure Prediction","date":"2022-11-27","arxiv_id":"2211.14939","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/applying-deep-reinforcement-learning-to-the#ran","syntology_url":"https://syntology.ai/paper/2211.14939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.14939"}},"official":{"repos":["compsoftmatterbiophysics-cityu-hk/applying-drl-to-hp-model-for-protein-structure-prediction"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bear-physics-principled-building-environment","slug":"bear-physics-principled-building-environment","title":"BEAR: Physics-Principled Building Environment for Control and Reinforcement Learning","date":"2022-11-27","arxiv_id":"2211.14744","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-and-safe-reinforcement-learning","slug":"explainable-and-safe-reinforcement-learning","title":"Explainable and Safe Reinforcement Learning for Autonomous Air Mobility","date":"2022-11-24","arxiv_id":"2211.13474","repositories_listed":1,"syntology":null},{"url":"/paper/actively-learning-costly-reward-functions-for","slug":"actively-learning-costly-reward-functions-for","title":"Actively Learning Costly Reward Functions for Reinforcement Learning","date":"2022-11-23","arxiv_id":"2211.13260","repositories_listed":1,"syntology":null},{"url":"/paper/masked-autoencoding-for-scalable-and","slug":"masked-autoencoding-for-scalable-and","title":"Masked Autoencoding for Scalable and Generalizable Decision Making","date":"2022-11-23","arxiv_id":"2211.12740","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/masked-autoencoding-for-scalable-and#ran","syntology_url":"https://syntology.ai/paper/2211.12740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12740"}},"official":{"repos":["fangchenliu/maskdp_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-low-latency-adaptive-coding-spiking","slug":"a-low-latency-adaptive-coding-spiking","title":"A Low Latency Adaptive Coding Spiking Framework for Deep Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11760","repositories_listed":1,"syntology":null},{"url":"/paper/examining-policy-entropy-of-reinforcement","slug":"examining-policy-entropy-of-reinforcement","title":"Examining Policy Entropy of Reinforcement Learning Agents for Personalization Tasks","date":"2022-11-21","arxiv_id":"2211.11869","repositories_listed":1,"syntology":null},{"url":"/paper/tempera-test-time-prompting-via-reinforcement","slug":"tempera-test-time-prompting-via-reinforcement","title":"TEMPERA: Test-Time Prompting via Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11890","repositories_listed":1,"syntology":null},{"url":"/paper/tinyqmix-distributed-access-control-for-mmtc","slug":"tinyqmix-distributed-access-control-for-mmtc","title":"TinyQMIX: Distributed Access Control for mMTC via Multi-agent Reinforcement Learning","date":"2022-11-21","arxiv_id":"2211.11692","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-meta-reinforcement-learning-for","slug":"efficient-meta-reinforcement-learning-for","title":"Efficient Meta Reinforcement Learning for Preference-based Fast Adaptation","date":"2022-11-20","arxiv_id":"2211.10861","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-meta-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10861"}},"official":{"repos":["stilwell-git/adaptation-with-noisy-oracle"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-search-for-job-shop-scheduling","slug":"learning-to-search-for-job-shop-scheduling","title":"Deep Reinforcement Learning Guided Improvement Heuristic for Job Shop Scheduling","date":"2022-11-20","arxiv_id":"2211.10936","repositories_listed":1,"syntology":null},{"url":"/paper/safelight-a-reinforcement-learning-method","slug":"safelight-a-reinforcement-learning-method","title":"SafeLight: A Reinforcement Learning Method toward Collision-free Traffic Signal Control","date":"2022-11-20","arxiv_id":"2211.10871","repositories_listed":1,"syntology":null},{"url":"/paper/reinform-selecting-paths-with-reinforcement","slug":"reinform-selecting-paths-with-reinforcement","title":"ReInform: Selecting paths with reinforcement learning for contextualized link prediction","date":"2022-11-19","arxiv_id":"2211.10688","repositories_listed":1,"syntology":null},{"url":"/paper/gosum-extractive-summarization-of-long","slug":"gosum-extractive-summarization-of-long","title":"GoSum: Extractive Summarization of Long Documents by Reinforcement Learning and Graph Organized discourse state","date":"2022-11-18","arxiv_id":"2211.10247","repositories_listed":1,"syntology":null},{"url":"/paper/language-conditioned-reinforcement-learning","slug":"language-conditioned-reinforcement-learning","title":"Language-Conditioned Reinforcement Learning to Solve Misunderstandings with Action Corrections","date":"2022-11-18","arxiv_id":"2211.10168","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-conditioned-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.10168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10168"}},"official":{"repos":["frankroeder/lanro-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-defense-against-backdoor-policies-in","slug":"provable-defense-against-backdoor-policies-in","title":"Provable Defense against Backdoor Policies in Reinforcement Learning","date":"2022-11-18","arxiv_id":"2211.10530","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/provable-defense-against-backdoor-policies-in#ran","syntology_url":"https://syntology.ai/paper/2211.10530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.10530"}},"official":{"repos":["skbharti/provable-defense-in-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-action-advising-for-multi-agent","slug":"explainable-action-advising-for-multi-agent","title":"Explainable Action Advising for Multi-Agent Reinforcement Learning","date":"2022-11-15","arxiv_id":"2211.07882","repositories_listed":1,"syntology":null},{"url":"/paper/interactively-learning-to-summarise-timelines","slug":"interactively-learning-to-summarise-timelines","title":"Towards Abstractive Timeline Summarisation using Preference-based Reinforcement Learning","date":"2022-11-14","arxiv_id":"2211.07596","repositories_listed":1,"syntology":null},{"url":"/paper/towards-data-driven-offline-simulations-for","slug":"towards-data-driven-offline-simulations-for","title":"Towards Data-Driven Offline Simulations for Online Reinforcement Learning","date":"2022-11-14","arxiv_id":"2211.07614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-data-driven-offline-simulations-for#ran","syntology_url":"https://syntology.ai/paper/2211.07614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07614"}},"official":{"repos":["microsoft/rl-offline-simulation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-heterogeneous-agent-cooperation-via","slug":"learning-heterogeneous-agent-cooperation-via","title":"Learning Heterogeneous Agent Cooperation via Multiagent League Training","date":"2022-11-13","arxiv_id":"2211.11616","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-explainable-reinforcement","slug":"a-survey-on-explainable-reinforcement","title":"A Survey on Explainable Reinforcement Learning: Concepts, Algorithms, Challenges","date":"2022-11-12","arxiv_id":"2211.06665","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-deep-reinforcement-learning-with-1","slug":"efficient-deep-reinforcement-learning-with-1","title":"Efficient Deep Reinforcement Learning with Predictive Processing Proximal Policy Optimization","date":"2022-11-11","arxiv_id":"2211.06236","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-sequentiality-in-reinforcement","slug":"leveraging-sequentiality-in-reinforcement","title":"Leveraging Sequentiality in Reinforcement Learning from a Single Demonstration","date":"2022-11-09","arxiv_id":"2211.04786","repositories_listed":1,"syntology":null},{"url":"/paper/doubly-inhomogeneous-reinforcement-learning","slug":"doubly-inhomogeneous-reinforcement-learning","title":"Doubly Inhomogeneous Reinforcement Learning","date":"2022-11-08","arxiv_id":"2211.03983","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-based-asymmetric-multi-task","slug":"curriculum-based-asymmetric-multi-task","title":"Curriculum-based Asymmetric Multi-task Reinforcement Learning","date":"2022-11-07","arxiv_id":"2211.03352","repositories_listed":1,"syntology":null},{"url":"/paper/design-process-is-a-reinforcement-learning","slug":"design-process-is-a-reinforcement-learning","title":"Design Process is a Reinforcement Learning Problem","date":"2022-11-06","arxiv_id":"2211.03136","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-quality-diversity-algorithms-on","slug":"benchmarking-quality-diversity-algorithms-on","title":"Benchmarking Quality-Diversity Algorithms on Neuroevolution for Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02193","repositories_listed":1,"syntology":null},{"url":"/paper/the-benefits-of-model-based-generalization-in","slug":"the-benefits-of-model-based-generalization-in","title":"The Benefits of Model-Based Generalization in Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02222","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-benefits-of-model-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2211.02222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02222"}},"official":{"repos":["kenjyoung/model_generalization_code_supplement"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-fully-observable-policies-for","slug":"leveraging-fully-observable-policies-for","title":"Leveraging Fully Observable Policies for Learning under Partial Observability","date":"2022-11-03","arxiv_id":"2211.01991","repositories_listed":1,"syntology":null},{"url":"/paper/synthesis-of-separation-processes-with","slug":"synthesis-of-separation-processes-with","title":"Synthesis of separation processes with reinforcement learning","date":"2022-11-03","arxiv_id":"2211.04327","repositories_listed":1,"syntology":null},{"url":"/paper/behavior-prior-representation-learning-for","slug":"behavior-prior-representation-learning-for","title":"Behavior Prior Representation learning for Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.00863","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/behavior-prior-representation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2211.00863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00863"}},"official":{"repos":["bit1029public/offline_bpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-13","slug":"multi-agent-reinforcement-learning-for-13","title":"Multi-Agent Reinforcement Learning for Adaptive Mesh Refinement","date":"2022-11-02","arxiv_id":"2211.00801","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-13#ran","syntology_url":"https://syntology.ai/paper/2211.00801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00801"}},"official":{"repos":["011235813/marl-amr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/spatial-temporal-recurrent-reinforcement","slug":"spatial-temporal-recurrent-reinforcement","title":"Spatial-temporal recurrent reinforcement learning for autonomous ships","date":"2022-11-02","arxiv_id":"2211.01004","repositories_listed":1,"syntology":null},{"url":"/paper/agent-time-attention-for-sparse-rewards-multi","slug":"agent-time-attention-for-sparse-rewards-multi","title":"Agent-Time Attention for Sparse Rewards Multi-Agent Reinforcement Learning","date":"2022-10-31","arxiv_id":"2210.17540","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-optimize-permutation-flow-shop","slug":"learning-to-optimize-permutation-flow-shop","title":"Learning to Optimize Permutation Flow Shop Scheduling via Graph-based Imitation Learning","date":"2022-10-31","arxiv_id":"2210.17178","repositories_listed":1,"syntology":null},{"url":"/paper/bimrl-brain-inspired-meta-reinforcement","slug":"bimrl-brain-inspired-meta-reinforcement","title":"BIMRL: Brain Inspired Meta Reinforcement Learning","date":"2022-10-29","arxiv_id":"2210.16530","repositories_listed":1,"syntology":null},{"url":"/paper/goal-exploration-augmentation-via-pre-trained","slug":"goal-exploration-augmentation-via-pre-trained","title":"Goal Exploration Augmentation via Pre-trained Skills for Sparse-Reward Long-Horizon Goal-Conditioned Reinforcement Learning","date":"2022-10-28","arxiv_id":"2210.16058","repositories_listed":1,"syntology":null},{"url":"/paper/environment-design-for-inverse-reinforcement","slug":"environment-design-for-inverse-reinforcement","title":"Environment Design for Inverse Reinforcement Learning","date":"2022-10-26","arxiv_id":"2210.14972","repositories_listed":1,"syntology":null},{"url":"/paper/low-rank-modular-reinforcement-learning-via","slug":"low-rank-modular-reinforcement-learning-via","title":"Low-Rank Modular Reinforcement Learning via Muscle Synergy","date":"2022-10-26","arxiv_id":"2210.15479","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/low-rank-modular-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2210.15479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15479"}},"official":{"repos":["drdh/synergy-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"b35575b8745c09571a49d41528feb3a93d58552fe7e8c486455c25b1d48459a1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}