{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/22","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":22,"pages_in_order":152,"rows_per_page":100,"rows":[2101,2200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/21","next":"/task/reinforcement-learning-1/papers/23","papers":[{"url":"/paper/non-zero-sum-game-control-for-multi-vehicle","slug":"non-zero-sum-game-control-for-multi-vehicle","title":"Non-zero-sum Game Control for Multi-vehicle Driving via Reinforcement Learning","date":"2023-02-08","arxiv_id":"2302.03958","repositories_listed":1,"syntology":null},{"url":"/paper/attacking-cooperative-multi-agent","slug":"attacking-cooperative-multi-agent","title":"Attacking Cooperative Multi-Agent Reinforcement Learning by Adversarial Minority Influence","date":"2023-02-07","arxiv_id":"2302.03322","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-recommendations-with-reinforcement","slug":"multi-task-recommendations-with-reinforcement","title":"Multi-Task Recommendations with Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03328","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-recommendations-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.03328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03328"}},"official":{"repos":["applied-machine-learning-lab/rmtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intrinsic-rewards-from-self-organizing","slug":"intrinsic-rewards-from-self-organizing","title":"Intrinsic Rewards from Self-Organizing Feature Maps for Exploration in Reinforcement Learning","date":"2023-02-06","arxiv_id":"2302.04125","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-quantum-gate-design-and","slug":"model-free-quantum-gate-design-and","title":"Model-free Quantum Gate Design and Calibration using Deep Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02371","repositories_listed":1,"syntology":null},{"url":"/paper/offline-learning-of-closed-loop-deep-brain","slug":"offline-learning-of-closed-loop-deep-brain","title":"Offline Learning of Closed-Loop Deep Brain Stimulation Controllers for Parkinson Disease Treatment","date":"2023-02-05","arxiv_id":"2302.02477","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-constrained-reinforcement","slug":"distributional-constrained-reinforcement","title":"Distributional constrained reinforcement learning for supply chain optimization","date":"2023-02-03","arxiv_id":"2302.01727","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-optimize-for-reinforcement","slug":"learning-to-optimize-for-reinforcement","title":"Learning to Optimize for Reinforcement Learning","date":"2023-02-03","arxiv_id":"2302.01470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-optimize-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2302.01470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01470"}},"official":{"repos":["sail-sg/optim4rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-gap-offline-policy-optimization-for","slug":"mind-the-gap-offline-policy-optimization-for","title":"Mind the Gap: Offline Policy Optimization for Imperfect Rewards","date":"2023-02-03","arxiv_id":"2302.01667","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/mind-the-gap-offline-policy-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2302.01667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01667"}},"official":{"repos":["Facebear-ljx/RGM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/two-stage-constrained-actor-critic-fo-short","slug":"two-stage-constrained-actor-critic-fo-short","title":"Two-Stage Constrained Actor-Critic for Short Video Recommendation","date":"2023-02-03","arxiv_id":"2302.01680","repositories_listed":1,"syntology":null},{"url":"/paper/policy-expansion-for-bridging-offline-to","slug":"policy-expansion-for-bridging-offline-to","title":"Policy Expansion for Bridging Offline-to-Online Reinforcement Learning","date":"2023-02-02","arxiv_id":"2302.00935","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-expansion-for-bridging-offline-to#ran","syntology_url":"https://syntology.ai/paper/2302.00935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00935"}},"official":{"repos":["haichao-zhang/pex"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/internally-rewarded-reinforcement-learning","slug":"internally-rewarded-reinforcement-learning","title":"Internally Rewarded Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00270","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internally-rewarded-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00270"}},"official":{"repos":["mengdi-li/internally-rewarded-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-3","slug":"a-reinforcement-learning-framework-for-3","title":"A Reinforcement Learning Framework for Dynamic Mediation Analysis","date":"2023-01-31","arxiv_id":"2301.13348","repositories_listed":1,"syntology":null},{"url":"/paper/crc-rl-a-novel-visual-feature-representation","slug":"crc-rl-a-novel-visual-feature-representation","title":"CRC-RL: A Novel Visual Feature Representation Architecture for Unsupervised Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13473","repositories_listed":1,"syntology":null},{"url":"/paper/execution-based-code-generation-using-deep","slug":"execution-based-code-generation-using-deep","title":"Execution-based Code Generation using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13816","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/execution-based-code-generation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2301.13816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13816"}},"official":{"repos":["reddy-lab-code-research/PPOCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-image-to-semantics-translation-for","slug":"few-shot-image-to-semantics-translation-for","title":"Few-Shot Image-to-Semantics Translation for Policy Transfer in Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13343","repositories_listed":1,"syntology":null},{"url":"/paper/learning-fast-and-slow-a-goal-directed-memory","slug":"learning-fast-and-slow-a-goal-directed-memory","title":"Learning, Fast and Slow: A Goal-Directed Memory-Based Approach for Dynamic Environments","date":"2023-01-31","arxiv_id":"2301.13758","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-transport-perturbations-for-safe","slug":"optimal-transport-perturbations-for-safe","title":"Optimal Transport Perturbations for Safe Reinforcement Learning with Robustness Guarantees","date":"2023-01-31","arxiv_id":"2301.13375","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-ddpm-sampling-with-shortcut-fine","slug":"optimizing-ddpm-sampling-with-shortcut-fine","title":"Optimizing DDPM Sampling with Shortcut Fine-Tuning","date":"2023-01-31","arxiv_id":"2301.13362","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/optimizing-ddpm-sampling-with-shortcut-fine#ran","syntology_url":"https://syntology.ai/paper/2301.13362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13362"}},"official":{"repos":["uw-madison-lee-lab/sft-pg"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/retrosynthetic-planning-with-dual-value","slug":"retrosynthetic-planning-with-dual-value","title":"Retrosynthetic Planning with Dual Value Networks","date":"2023-01-31","arxiv_id":"2301.13755","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/retrosynthetic-planning-with-dual-value#ran","syntology_url":"https://syntology.ai/paper/2301.13755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13755"}},"official":{"repos":["DiXue98/PDVN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-online-reinforcement-learning-with","slug":"guiding-online-reinforcement-learning-with","title":"Guiding Online Reinforcement Learning with Action-Free Offline Pretraining","date":"2023-01-30","arxiv_id":"2301.12876","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guiding-online-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2301.12876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12876"}},"official":{"repos":["vision-cair/af-guide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/importance-weighted-actor-critic-for-optimal-1","slug":"importance-weighted-actor-critic-for-optimal-1","title":"Importance Weighted Actor-Critic for Optimal Conservative Offline Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.12714","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/importance-weighted-actor-critic-for-optimal-1#ran","syntology_url":"https://syntology.ai/paper/2301.12714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12714"}},"official":{"repos":["zhuhl98/acrab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pac-bayesian-soft-actor-critic-learning","slug":"pac-bayesian-soft-actor-critic-learning","title":"PAC-Bayesian Soft Actor-Critic Learning","date":"2023-01-30","arxiv_id":"2301.12776","repositories_listed":1,"syntology":null},{"url":"/paper/planning-multiple-epidemic-interventions-with","slug":"planning-multiple-epidemic-interventions-with","title":"Planning Multiple Epidemic Interventions with Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.12802","repositories_listed":1,"syntology":null},{"url":"/paper/winning-solution-of-real-robot-challenge-iii","slug":"winning-solution-of-real-robot-challenge-iii","title":"Identifying Expert Behavior in Offline Training Datasets Improves Behavioral Cloning of Robotic Manipulation Policies","date":"2023-01-30","arxiv_id":"2301.13019","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/winning-solution-of-real-robot-challenge-iii#ran","syntology_url":"https://syntology.ai/paper/2301.13019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13019"}},"official":{"repos":["wq13552463699/real-robot-challenge-2022"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/apac-authorized-probability-controlled-actor","slug":"apac-authorized-probability-controlled-actor","title":"Constrained Policy Optimization with Explicit Behavior Density for Offline Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12130","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/apac-authorized-probability-controlled-actor#ran","syntology_url":"https://syntology.ai/paper/2301.12130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12130"}},"official":{"repos":["evalarzj/cped"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-embodied-agents-dream-of-pixelated-sheep","slug":"do-embodied-agents-dream-of-pixelated-sheep","title":"Do Embodied Agents Dream of Pixelated Sheep: Embodied Decision Making using Language Guided World Modelling","date":"2023-01-28","arxiv_id":"2301.12050","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/do-embodied-agents-dream-of-pixelated-sheep#ran","syntology_url":"https://syntology.ai/paper/2301.12050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12050"}},"official":null}},{"url":"/paper/outcome-directed-reinforcement-learning-by","slug":"outcome-directed-reinforcement-learning-by","title":"Outcome-directed Reinforcement Learning by Uncertainty & Temporal Distance-Aware Curriculum Goal Generation","date":"2023-01-27","arxiv_id":"2301.11741","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":1,"n_ran_checked":2,"n_instrument":5,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/outcome-directed-reinforcement-learning-by#ran","syntology_url":"https://syntology.ai/paper/2301.11741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11741"}},"official":{"repos":["jaylee0301/outpace_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/automatic-intrinsic-reward-shaping-for","slug":"automatic-intrinsic-reward-shaping-for","title":"Automatic Intrinsic Reward Shaping for Exploration in Deep Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.10886","repositories_listed":1,"syntology":null},{"url":"/paper/deep-laplacian-based-options-for-temporally","slug":"deep-laplacian-based-options-for-temporally","title":"Deep Laplacian-based Options for Temporally-Extended Exploration","date":"2023-01-26","arxiv_id":"2301.11181","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-laplacian-based-options-for-temporally#ran","syntology_url":"https://syntology.ai/paper/2301.11181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11181"}},"official":{"repos":["mklissa/dceo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-trust-region-based-safe","slug":"efficient-trust-region-based-safe","title":"Trust Region-Based Safe Distributional Reinforcement Learning for Multiple Constraints","date":"2023-01-26","arxiv_id":"2301.10923","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-trust-region-based-safe#ran","syntology_url":"https://syntology.ai/paper/2301.10923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.10923"}},"official":{"repos":["rllab-snu/safe-distributional-actor-critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-multiple-independent-advisors","slug":"learning-from-multiple-independent-advisors","title":"Learning from Multiple Independent Advisors in Multi-agent Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11153","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-aware-eligibility-traces-for-off","slug":"trajectory-aware-eligibility-traces-for-off","title":"Trajectory-Aware Eligibility Traces for Off-Policy Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.11321","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trajectory-aware-eligibility-traces-for-off#ran","syntology_url":"https://syntology.ai/paper/2301.11321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11321"}},"official":{"repos":["brett-daley/trajectory-aware-etraces"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/which-experiences-are-influential-for-your","slug":"which-experiences-are-influential-for-your","title":"Which Experiences Are Influential for Your Agent? Policy Iteration with Turn-over Dropout","date":"2023-01-26","arxiv_id":"2301.11168","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-control-of-partial-differential","slug":"distributed-control-of-partial-differential","title":"Distributed Control of Partial Differential Equations Using Convolutional Reinforcement Learning","date":"2023-01-25","arxiv_id":"2301.10737","repositories_listed":1,"syntology":null},{"url":"/paper/select-and-trade-towards-unified-pair-trading","slug":"select-and-trade-towards-unified-pair-trading","title":"Select and Trade: Towards Unified Pair Trading with Hierarchical Reinforcement Learning","date":"2023-01-25","arxiv_id":"2301.10724","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-reinforcement-learning-for-2","slug":"constrained-reinforcement-learning-for-2","title":"Constrained Reinforcement Learning for Dexterous Manipulation","date":"2023-01-24","arxiv_id":"2301.09766","repositories_listed":1,"syntology":null},{"url":"/paper/the-configurable-tree-graph-ct-graph","slug":"the-configurable-tree-graph-ct-graph","title":"The configurable tree graph (CT-graph): measurable problems in partially observable and distal reward environments for lifelong reinforcement learning","date":"2023-01-21","arxiv_id":"2302.10887","repositories_listed":1,"syntology":null},{"url":"/paper/pirlnav-pretraining-with-imitation-and-rl","slug":"pirlnav-pretraining-with-imitation-and-rl","title":"PIRLNav: Pretraining with Imitation and RL Finetuning for ObjectNav","date":"2023-01-18","arxiv_id":"2301.07302","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-path-planning","slug":"a-reinforcement-learning-path-planning","title":"A reinforcement learning path planning approach for range-only underwater target localization with autonomous vehicles","date":"2023-01-17","arxiv_id":"2301.06863","repositories_listed":1,"syntology":null},{"url":"/paper/the-swannflight-system-on-the-fly-sim-to-real","slug":"the-swannflight-system-on-the-fly-sim-to-real","title":"Sim-Anchored Learning for On-the-Fly Adaptation","date":"2023-01-17","arxiv_id":"2301.06987","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-path","slug":"deep-reinforcement-learning-based-path","title":"Deep-Reinforcement-Learning-based Path Planning for Industrial Robots using Distance Sensors as Observation","date":"2023-01-14","arxiv_id":"2301.05980","repositories_listed":1,"syntology":null},{"url":"/paper/mutation-testing-of-deep-reinforcement","slug":"mutation-testing-of-deep-reinforcement","title":"Mutation Testing of Deep Reinforcement Learning Based on Real Faults","date":"2023-01-13","arxiv_id":"2301.05651","repositories_listed":1,"syntology":null},{"url":"/paper/predictive-world-models-from-real-world","slug":"predictive-world-models-from-real-world","title":"Predictive World Models from Real-World Partial Observations","date":"2023-01-12","arxiv_id":"2301.04783","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-online-multi-task-reinforcement","slug":"adversarial-online-multi-task-reinforcement","title":"Adversarial Online Multi-Task Reinforcement Learning","date":"2023-01-11","arxiv_id":"2301.04268","repositories_listed":1,"syntology":null},{"url":"/paper/hint-assisted-reinforcement-learning-an","slug":"hint-assisted-reinforcement-learning-an","title":"Hint assisted reinforcement learning: an application in radio astronomy","date":"2023-01-10","arxiv_id":"2301.03933","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-perceive-in-deep-model-free","slug":"learning-to-perceive-in-deep-model-free","title":"Learning to Perceive in Deep Model-Free Reinforcement Learning","date":"2023-01-10","arxiv_id":"2301.03730","repositories_listed":1,"syntology":null},{"url":"/paper/orbit-a-unified-simulation-framework-for","slug":"orbit-a-unified-simulation-framework-for","title":"Orbit: A Unified Simulation Framework for Interactive Robot Learning Environments","date":"2023-01-10","arxiv_id":"2301.04195","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/orbit-a-unified-simulation-framework-for#ran","syntology_url":"https://syntology.ai/paper/2301.04195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.04195"}},"official":{"repos":["NVIDIA-Omniverse/Orbit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/schlably-a-python-framework-for-deep","slug":"schlably-a-python-framework-for-deep","title":"schlably: A Python Framework for Deep Reinforcement Learning Based Scheduling Experiments","date":"2023-01-10","arxiv_id":"2301.04182","repositories_listed":1,"syntology":null},{"url":"/paper/why-people-skip-music-on-predicting-music","slug":"why-people-skip-music-on-predicting-music","title":"Why People Skip Music? On Predicting Music Skips using Deep Reinforcement Learning","date":"2023-01-10","arxiv_id":"2301.03881","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-collective-intelligence-from-massive","slug":"emergent-collective-intelligence-from-massive","title":"Emergent collective intelligence from massive-agent cooperation and competition","date":"2023-01-04","arxiv_id":"2301.01609","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergent-collective-intelligence-from-massive#ran","syntology_url":"https://syntology.ai/paper/2301.01609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.01609"}},"official":{"repos":["hanmochen/lux-open"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robofriend-an-adpative-storytelling-robotic","slug":"robofriend-an-adpative-storytelling-robotic","title":"Robofriend: An Adpative Storytelling Robotic Teddy Bear - Technical Report","date":"2023-01-04","arxiv_id":"2301.01576","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-irrigation","slug":"deep-reinforcement-learning-for-irrigation","title":"Deep reinforcement learning for irrigation scheduling using high-dimensional sensor feedback","date":"2023-01-02","arxiv_id":"2301.00899","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-maximize-mutual-information-for","slug":"learning-to-maximize-mutual-information-for","title":"Learning to Maximize Mutual Information for Dynamic Feature Selection","date":"2023-01-02","arxiv_id":"2301.00557","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-maximize-mutual-information-for#ran","syntology_url":"https://syntology.ai/paper/2301.00557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00557"}},"official":{"repos":["iancovert/dynamic-selection"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-challenges-of-using-reinforcement","slug":"on-the-challenges-of-using-reinforcement","title":"On the Challenges of using Reinforcement Learning in Precision Drug Dosing: Delay and Prolongedness of Action Effects","date":"2023-01-02","arxiv_id":"2301.00512","repositories_listed":1,"syntology":null},{"url":"/paper/environment-agnostic-representation-for","slug":"environment-agnostic-representation-for","title":"Environment Agnostic Representation for Visual Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/goal-guided-transformer-enabled-reinforcement","slug":"goal-guided-transformer-enabled-reinforcement","title":"Goal-Guided Transformer-Enabled Reinforcement Learning for Efficient Autonomous Navigation","date":"2023-01-01","arxiv_id":"2301.00362","repositories_listed":1,"syntology":null},{"url":"/paper/self-activating-neural-ensembles-for-1","slug":"self-activating-neural-ensembles-for-1","title":"Self-Activating Neural Ensembles for Continual Reinforcement Learning","date":"2022-12-31","arxiv_id":"2301.00141","repositories_listed":1,"syntology":null},{"url":"/paper/pontryagin-optimal-controller-via-neural","slug":"pontryagin-optimal-controller-via-neural","title":"Pontryagin Optimal Control via Neural Networks","date":"2022-12-30","arxiv_id":"2212.14566","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-success-induced","slug":"reinforcement-learning-with-success-induced","title":"Reinforcement Learning with Success Induced Task Prioritization","date":"2022-12-30","arxiv_id":"2301.00691","repositories_listed":1,"syntology":null},{"url":"/paper/risk-sensitive-policy-with-distributional","slug":"risk-sensitive-policy-with-distributional","title":"Risk-Sensitive Policy with Distributional Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14743","repositories_listed":1,"syntology":null},{"url":"/paper/rl-and-fingerprinting-to-select-moving-target","slug":"rl-and-fingerprinting-to-select-moving-target","title":"RL and Fingerprinting to Select Moving Target Defense Mechanisms for Zero-day Attacks in IoT","date":"2022-12-30","arxiv_id":"2212.14647","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-visual-reinforcement-learning-a","slug":"symbolic-visual-reinforcement-learning-a","title":"Symbolic Visual Reinforcement Learning: A Scalable Framework with Object-Level Abstraction and Differentiable Expression Search","date":"2022-12-30","arxiv_id":"2212.14849","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-in-transformer-as-backbone-for","slug":"transformer-in-transformer-as-backbone-for","title":"Transformer in Transformer as Backbone for Deep Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14538","repositories_listed":1,"syntology":null},{"url":"/paper/lexicographic-multi-objective-reinforcement","slug":"lexicographic-multi-objective-reinforcement","title":"Lexicographic Multi-Objective Reinforcement Learning","date":"2022-12-28","arxiv_id":"2212.13769","repositories_listed":1,"syntology":null},{"url":"/paper/on-pathologies-in-kl-regularized-1","slug":"on-pathologies-in-kl-regularized-1","title":"On Pathologies in KL-Regularized Reinforcement Learning from Expert Demonstrations","date":"2022-12-28","arxiv_id":"2212.13936","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-pathologies-in-kl-regularized-1#ran","syntology_url":"https://syntology.ai/paper/2212.13936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13936"}},"official":{"repos":["conglu1997/nppac"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/strangeness-driven-exploration-in-multi-agent","slug":"strangeness-driven-exploration-in-multi-agent","title":"Strangeness-driven Exploration in Multi-Agent Reinforcement Learning","date":"2022-12-27","arxiv_id":"2212.13448","repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-representations-for-1","slug":"learning-generalizable-representations-for-1","title":"Learning Generalizable Representations for Reinforcement Learning via Adaptive Meta-learner of Behavioral Similarities","date":"2022-12-26","arxiv_id":"2212.13088","repositories_listed":1,"syntology":null},{"url":"/paper/example-guided-learning-of-stochastic-human","slug":"example-guided-learning-of-stochastic-human","title":"Example-guided learning of stochastic human driving policies using deep reinforcement learning","date":"2022-12-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/nars-vs-reinforcement-learning-ona-vs-q","slug":"nars-vs-reinforcement-learning-ona-vs-q","title":"NARS vs. Reinforcement learning: ONA vs. Q-Learning","date":"2022-12-23","arxiv_id":"2212.12517","repositories_listed":1,"syntology":null},{"url":"/paper/certified-policy-smoothing-for-cooperative","slug":"certified-policy-smoothing-for-cooperative","title":"Certified Policy Smoothing for Cooperative Multi-Agent Reinforcement Learning","date":"2022-12-22","arxiv_id":"2212.11746","repositories_listed":1,"syntology":null},{"url":"/paper/control-of-continuous-quantum-systems-with","slug":"control-of-continuous-quantum-systems-with","title":"Control of Continuous Quantum Systems with Many Degrees of Freedom based on Convergent Reinforcement Learning","date":"2022-12-21","arxiv_id":"2212.10705","repositories_listed":1,"syntology":null},{"url":"/paper/critic-guided-decoding-for-controlled-text","slug":"critic-guided-decoding-for-controlled-text","title":"Critic-Guided Decoding for Controlled Text Generation","date":"2022-12-21","arxiv_id":"2212.10938","repositories_listed":1,"syntology":null},{"url":"/paper/generating-multiple-length-summaries-via","slug":"generating-multiple-length-summaries-via","title":"Generating Multiple-Length Summaries via Reinforcement Learning for Unsupervised Sentence Summarization","date":"2022-12-21","arxiv_id":"2212.10843","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameters-in-contextual-rl-are-highly","slug":"hyperparameters-in-contextual-rl-are-highly","title":"Hyperparameters in Contextual RL are Highly Situational","date":"2022-12-21","arxiv_id":"2212.10876","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameters-in-contextual-rl-are-highly#ran","syntology_url":"https://syntology.ai/paper/2212.10876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10876"}},"official":{"repos":["automl-private/crl_hpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-reinforcement-learning-for-the-game-of","slug":"on-reinforcement-learning-for-the-game-of","title":"On Reinforcement Learning for the Game of 2048","date":"2022-12-21","arxiv_id":"2212.11087","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-of-model-free-and-model-based","slug":"comparison-of-model-free-and-model-based","title":"Comparison of Model-Free and Model-Based Learning-Informed Planning for PointGoal Navigation","date":"2022-12-17","arxiv_id":"2212.08801","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-for-visual","slug":"offline-reinforcement-learning-for-visual","title":"Offline Reinforcement Learning for Visual Navigation","date":"2022-12-16","arxiv_id":"2212.08244","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-training-and-execution-multi","slug":"distributed-training-and-execution-multi","title":"Distributed-Training-and-Execution Multi-Agent Reinforcement Learning for Power Control in HetNet","date":"2022-12-15","arxiv_id":"2212.07967","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-multi-agent-deep-reinforcement","slug":"hybrid-multi-agent-deep-reinforcement","title":"Hybrid Multi-agent Deep Reinforcement Learning for Autonomous Mobility on Demand Systems","date":"2022-12-14","arxiv_id":"2212.07313","repositories_listed":1,"syntology":null},{"url":"/paper/robust-policy-optimization-in-deep","slug":"robust-policy-optimization-in-deep","title":"Robust Policy Optimization in Deep Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.07536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-policy-optimization-in-deep#ran","syntology_url":"https://syntology.ai/paper/2212.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07536"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-and-tree-search","slug":"reinforcement-learning-and-tree-search","title":"Reinforcement Learning and Tree Search Methods for the Unit Commitment Problem","date":"2022-12-12","arxiv_id":"2212.06001","repositories_listed":1,"syntology":null},{"url":"/paper/effects-of-spectral-normalization-in-multi","slug":"effects-of-spectral-normalization-in-multi","title":"Effects of Spectral Normalization in Multi-agent Reinforcement Learning","date":"2022-12-10","arxiv_id":"2212.05331","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effects-of-spectral-normalization-in-multi#ran","syntology_url":"https://syntology.ai/paper/2212.05331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05331"}},"official":{"repos":["kinalmehta/epymarl_spectral"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-risk-aware-bidding-with-budget","slug":"adaptive-risk-aware-bidding-with-budget","title":"Adaptive Risk-Aware Bidding with Budget Constraint in Display Advertising","date":"2022-12-06","arxiv_id":"2212.12533","repositories_listed":1,"syntology":null},{"url":"/paper/prefrec-preference-based-recommender-systems","slug":"prefrec-preference-based-recommender-systems","title":"PrefRec: Recommender Systems with Human Preferences for Reinforcing Long-term User Engagement","date":"2022-12-06","arxiv_id":"2212.02779","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/prefrec-preference-based-recommender-systems#ran","syntology_url":"https://syntology.ai/paper/2212.02779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02779"}},"official":null}},{"url":"/paper/solving-the-side-chain-packing-arrangement-of","slug":"solving-the-side-chain-packing-arrangement-of","title":"Reinforcement Learning for Molecular Dynamics Optimization: A Stochastic Pontryagin Maximum Principle Approach","date":"2022-12-06","arxiv_id":"2212.03320","repositories_listed":1,"syntology":null},{"url":"/paper/state-space-closure-revisiting-endless-online","slug":"state-space-closure-revisiting-endless-online","title":"State Space Closure: Revisiting Endless Online Level Generation via Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.02951","repositories_listed":1,"syntology":null},{"url":"/paper/switching-to-discriminative-image-captioning","slug":"switching-to-discriminative-image-captioning","title":"Switching to Discriminative Image Captioning by Relieving a Bottleneck of Reinforcement Learning","date":"2022-12-06","arxiv_id":"2212.03230","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-the-solution-for-state-adversarial","slug":"what-is-the-solution-for-state-adversarial","title":"What is the Solution for State-Adversarial Multi-Agent Reinforcement Learning?","date":"2022-12-06","arxiv_id":"2212.02705","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-is-the-solution-for-state-adversarial#ran","syntology_url":"https://syntology.ai/paper/2212.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02705"}},"official":{"repos":["susanbao/rmarl_code"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-machine-with-short-term-episodic-and","slug":"a-machine-with-short-term-episodic-and","title":"A Machine with Short-Term, Episodic, and Semantic Memory Systems","date":"2022-12-05","arxiv_id":"2212.02098","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-machine-with-short-term-episodic-and#ran","syntology_url":"https://syntology.ai/paper/2212.02098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02098"}},"official":{"repos":["humemai/agent-room-env-v1"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/l2sr-learning-to-sample-and-reconstruct-for","slug":"l2sr-learning-to-sample-and-reconstruct-for","title":"L2SR: Learning to Sample and Reconstruct for Accelerated MRI via Reinforcement Learning","date":"2022-12-05","arxiv_id":"2212.02190","repositories_listed":1,"syntology":null},{"url":"/paper/learning-physically-realizable-skills-for","slug":"learning-physically-realizable-skills-for","title":"Learning Physically Realizable Skills for Online Packing of General 3D Shapes","date":"2022-12-05","arxiv_id":"2212.02094","repositories_listed":1,"syntology":null},{"url":"/paper/physics-informed-model-based-reinforcement","slug":"physics-informed-model-based-reinforcement","title":"Physics-Informed Model-Based Reinforcement Learning","date":"2022-12-05","arxiv_id":"2212.02179","repositories_listed":1,"syntology":null},{"url":"/paper/td3-with-reverse-kl-regularizer-for-offline","slug":"td3-with-reverse-kl-regularizer-for-offline","title":"TD3 with Reverse KL Regularizer for Offline Reinforcement Learning from Mixed Datasets","date":"2022-12-05","arxiv_id":"2212.02125","repositories_listed":1,"syntology":null},{"url":"/paper/automata-learning-meets-shielding","slug":"automata-learning-meets-shielding","title":"Automata Learning meets Shielding","date":"2022-12-04","arxiv_id":"2212.01838","repositories_listed":1,"syntology":null},{"url":"/paper/rlogist-fast-observation-strategy-on-whole","slug":"rlogist-fast-observation-strategy-on-whole","title":"RLogist: Fast Observation Strategy on Whole-slide Images with Deep Reinforcement Learning","date":"2022-12-04","arxiv_id":"2212.01737","repositories_listed":1,"syntology":null},{"url":"/paper/meshdqn-a-deep-reinforcement-learning","slug":"meshdqn-a-deep-reinforcement-learning","title":"MeshDQN: A Deep Reinforcement Learning Framework for Improving Meshes in Computational Fluid Dynamics","date":"2022-12-02","arxiv_id":"2212.01428","repositories_listed":1,"syntology":null},{"url":"/paper/stl-based-synthesis-of-feedback-controllers","slug":"stl-based-synthesis-of-feedback-controllers","title":"STL-Based Synthesis of Feedback Controllers Using Reinforcement Learning","date":"2022-12-02","arxiv_id":"2212.01022","repositories_listed":1,"syntology":null},{"url":"/paper/karolos-an-open-source-reinforcement-learning","slug":"karolos-an-open-source-reinforcement-learning","title":"Karolos: An Open-Source Reinforcement Learning Framework for Robot-Task Environments","date":"2022-12-01","arxiv_id":"2212.00906","repositories_listed":1,"syntology":null}],"record_sha256":"47ce3f5f70c7635e8fd6829d895a36266137ced9cb80cede110556ad14e29adf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}