{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/18","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":135,"rows_per_page":100,"rows":[1701,1800],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/17","next":"/task/reinforcement-learning-2/papers/19","papers":[{"url":"/paper/end-to-end-reinforcement-learning-for-torque","slug":"end-to-end-reinforcement-learning-for-torque","title":"End-to-End Reinforcement Learning for Torque Based Variable Height Hopping","date":"2023-07-31","arxiv_id":"2307.16676","repositories_listed":1,"syntology":null},{"url":"/paper/value-informed-skill-chaining-for-policy","slug":"value-informed-skill-chaining-for-policy","title":"Value-Informed Skill Chaining for Policy Learning of Long-Horizon Tasks with Surgical Robot","date":"2023-07-31","arxiv_id":"2307.16503","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/value-informed-skill-chaining-for-policy#ran","syntology_url":"https://syntology.ai/paper/2307.16503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16503"}},"official":{"repos":["med-air/viskill"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/drl4route-a-deep-reinforcement-learning","slug":"drl4route-a-deep-reinforcement-learning","title":"DRL4Route: A Deep Reinforcement Learning Framework for Pick-up and Delivery Route Prediction","date":"2023-07-30","arxiv_id":"2307.16246","repositories_listed":1,"syntology":null},{"url":"/paper/robust-multi-agent-reinforcement-learning-3","slug":"robust-multi-agent-reinforcement-learning-3","title":"Robust Multi-Agent Reinforcement Learning with State Uncertainty","date":"2023-07-30","arxiv_id":"2307.16212","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2307.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16212"}},"official":{"repos":["sihongho/robust_marl_with_state_uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/variance-control-for-distributional","slug":"variance-control-for-distributional","title":"Variance Control for Distributional Reinforcement Learning","date":"2023-07-30","arxiv_id":"2307.16152","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-driven-reinforcement-learning-based","slug":"curiosity-driven-reinforcement-learning-based","title":"Curiosity-Driven Reinforcement Learning based Low-Level Flight Control","date":"2023-07-28","arxiv_id":"2307.15724","repositories_listed":1,"syntology":null},{"url":"/paper/worrisome-properties-of-neural-network","slug":"worrisome-properties-of-neural-network","title":"Worrisome Properties of Neural Network Controllers and Their Symbolic Representations","date":"2023-07-28","arxiv_id":"2307.15456","repositories_listed":1,"syntology":null},{"url":"/paper/approximate-model-based-shielding-for-safe","slug":"approximate-model-based-shielding-for-safe","title":"Approximate Model-Based Shielding for Safe Reinforcement Learning","date":"2023-07-27","arxiv_id":"2308.00707","repositories_listed":1,"syntology":null},{"url":"/paper/flare-fingerprinting-deep-reinforcement","slug":"flare-fingerprinting-deep-reinforcement","title":"FLARE: Fingerprinting Deep Reinforcement Learning Agents using Universal Adversarial Masks","date":"2023-07-27","arxiv_id":"2307.14751","repositories_listed":1,"syntology":null},{"url":"/paper/2-level-reinforcement-learning-for-ships-on","slug":"2-level-reinforcement-learning-for-ships-on","title":"2-Level Reinforcement Learning for Ships on Inland Waterways: Path Planning and Following","date":"2023-07-25","arxiv_id":"2307.16769","repositories_listed":1,"syntology":null},{"url":"/paper/mode-constrained-model-based-reinforcement","slug":"mode-constrained-model-based-reinforcement","title":"Mode-constrained Model-based Reinforcement Learning via Gaussian Processes","date":"2023-07-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-adaptation-and","slug":"reinforcement-learning-based-adaptation-and","title":"Reinforcement Learning -based Adaptation and Scheduling Methods for Multi-source DASH","date":"2023-07-25","arxiv_id":"2308.11621","repositories_listed":1,"syntology":null},{"url":"/paper/submodular-reinforcement-learning","slug":"submodular-reinforcement-learning","title":"Submodular Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13372","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/submodular-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.13372","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.13372"}},"official":{"repos":["manish-pra/non-additive-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-connection-between-one-step-regularization","slug":"a-connection-between-one-step-regularization","title":"A Connection between One-Step Regularization and Critic Regularization in Reinforcement Learning","date":"2023-07-24","arxiv_id":"2307.12968","repositories_listed":1,"syntology":null},{"url":"/paper/policy-gradient-optimal-correlation-search","slug":"policy-gradient-optimal-correlation-search","title":"Policy Gradient Optimal Correlation Search for Variance Reduction in Monte Carlo simulation and Maximum Optimal Transport","date":"2023-07-24","arxiv_id":"2307.12703","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-effectiveness-of-offline-rl-for","slug":"on-the-effectiveness-of-offline-rl-for","title":"On the Effectiveness of Offline RL for Dialogue Response Generation","date":"2023-07-23","arxiv_id":"2307.12425","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/on-the-effectiveness-of-offline-rl-for#ran","syntology_url":"https://syntology.ai/paper/2307.12425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12425"}},"official":{"repos":["asappresearch/dialogue-offline-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/balancing-exploration-and-exploitation-in","slug":"balancing-exploration-and-exploitation-in","title":"Balancing Exploration and Exploitation in Hierarchical Reinforcement Learning via Latent Landmark Graphs","date":"2023-07-22","arxiv_id":"2307.12063","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/balancing-exploration-and-exploitation-in#ran","syntology_url":"https://syntology.ai/paper/2307.12063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12063"}},"official":{"repos":["papercode2022/hill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emergence-of-adaptive-circadian-rhythms-in","slug":"emergence-of-adaptive-circadian-rhythms-in","title":"Emergence of Adaptive Circadian Rhythms in Deep Reinforcement Learning","date":"2023-07-22","arxiv_id":"2307.12143","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergence-of-adaptive-circadian-rhythms-in#ran","syntology_url":"https://syntology.ai/paper/2307.12143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12143"}},"official":{"repos":["aqeel13932/mn_project"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hindsight-dice-stable-credit-assignment-for","slug":"hindsight-dice-stable-credit-assignment-for","title":"Hindsight-DICE: Stable Credit Assignment for Deep Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11897","repositories_listed":1,"syntology":null},{"url":"/paper/joingym-an-efficient-query-optimization","slug":"joingym-an-efficient-query-optimization","title":"JoinGym: An Efficient Query Optimization Environment for Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11704","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-offline-reinforcement-learning-2","slug":"model-based-offline-reinforcement-learning-2","title":"Model-based Offline Reinforcement Learning with Count-based Conservatism","date":"2023-07-21","arxiv_id":"2307.11352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/model-based-offline-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2307.11352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11352"}},"official":{"repos":["oh-lab/count-morl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/breadcrumbs-to-the-goal-goal-conditioned","slug":"breadcrumbs-to-the-goal-goal-conditioned","title":"Breadcrumbs to the Goal: Goal-Conditioned Exploration from Human-in-the-Loop Feedback","date":"2023-07-20","arxiv_id":"2307.11049","repositories_listed":1,"syntology":null},{"url":"/paper/longitudinal-data-and-a-semantic-similarity","slug":"longitudinal-data-and-a-semantic-similarity","title":"Longitudinal Data and a Semantic Similarity Reward for Chest X-Ray Report Generation","date":"2023-07-19","arxiv_id":"2307.09758","repositories_listed":1,"syntology":null},{"url":"/paper/pytag-challenges-and-opportunities-for","slug":"pytag-challenges-and-opportunities-for","title":"PyTAG: Challenges and Opportunities for Reinforcement Learning in Tabletop Games","date":"2023-07-19","arxiv_id":"2307.09905","repositories_listed":1,"syntology":null},{"url":"/paper/strapper-preference-based-reinforcement","slug":"strapper-preference-based-reinforcement","title":"STRAPPER: Preference-based Reinforcement Learning via Self-training Augmentation and Peer Regularization","date":"2023-07-19","arxiv_id":"2307.09692","repositories_listed":1,"syntology":null},{"url":"/paper/natural-actor-critic-for-robust-reinforcement","slug":"natural-actor-critic-for-robust-reinforcement","title":"Natural Actor-Critic for Robust Reinforcement Learning with Function Approximation","date":"2023-07-17","arxiv_id":"2307.08875","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/natural-actor-critic-for-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2307.08875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08875"}},"official":{"repos":["tliu1997/rnac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-vigen-a-reinforcement-learning-benchmark-1","slug":"rl-vigen-a-reinforcement-learning-benchmark-1","title":"RL-ViGen: A Reinforcement Learning Benchmark for Visual Generalization","date":"2023-07-15","arxiv_id":"2307.10224","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rl-vigen-a-reinforcement-learning-benchmark-1#ran","syntology_url":"https://syntology.ai/paper/2307.10224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10224"}},"official":{"repos":["gemcollector/rl-vigen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-photonic-component","slug":"reinforcement-learning-for-photonic-component","title":"Reinforcement Learning for Photonic Component Design","date":"2023-07-14","arxiv_id":"2307.11075","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-frontier-based","slug":"reinforcement-learning-with-frontier-based","title":"Reinforcement Learning with Frontier-Based Exploration via Autonomous Environment","date":"2023-07-14","arxiv_id":"2307.07296","repositories_listed":1,"syntology":null},{"url":"/paper/safe-dreamerv3-safe-reinforcement-learning","slug":"safe-dreamerv3-safe-reinforcement-learning","title":"SafeDreamer: Safe Reinforcement Learning with World Models","date":"2023-07-14","arxiv_id":"2307.07176","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-dreamerv3-safe-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.07176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07176"}},"official":{"repos":["pku-alignment/safedreamer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/defeating-proactive-jammers-using-deep","slug":"defeating-proactive-jammers-using-deep","title":"Defeating Proactive Jammers Using Deep Reinforcement Learning for Resource-Constrained IoT Networks","date":"2023-07-13","arxiv_id":"2307.06796","repositories_listed":1,"syntology":null},{"url":"/paper/prescriptive-process-monitoring-under-1","slug":"prescriptive-process-monitoring-under-1","title":"Prescriptive Process Monitoring Under Resource Constraints: A Reinforcement Learning Approach","date":"2023-07-13","arxiv_id":"2307.06564","repositories_listed":1,"syntology":null},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-as-wasserstein","slug":"safe-reinforcement-learning-as-wasserstein","title":"Probabilistic Constrained Reinforcement Learning with Formal Interpretability","date":"2023-07-13","arxiv_id":"2307.07084","repositories_listed":1,"syntology":null},{"url":"/paper/dsse-a-drone-swarm-search-environment","slug":"dsse-a-drone-swarm-search-environment","title":"DSSE: a drone swarm search environment","date":"2023-07-12","arxiv_id":"2307.06240","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-experimental-design-for-x-ray-ct","slug":"sequential-experimental-design-for-x-ray-ct","title":"Sequential Experimental Design for X-Ray CT Using Deep Reinforcement Learning","date":"2023-07-12","arxiv_id":"2307.06343","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-feedback-efficiency-of-interactive","slug":"boosting-feedback-efficiency-of-interactive","title":"Boosting Feedback Efficiency of Interactive Reinforcement Learning by Adaptive Learning from Scores","date":"2023-07-11","arxiv_id":"2307.05405","repositories_listed":1,"syntology":null},{"url":"/paper/empowering-recommender-systems-using","slug":"empowering-recommender-systems-using","title":"Empowering recommender systems using automatically generated Knowledge Graphs and Reinforcement Learning","date":"2023-07-11","arxiv_id":"2307.04996","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsically-motivated-graph-exploration","slug":"intrinsically-motivated-graph-exploration","title":"Intrinsically motivated graph exploration using network theories of human curiosity","date":"2023-07-11","arxiv_id":"2307.04962","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-non-cumulative","slug":"reinforcement-learning-with-non-cumulative","title":"Reinforcement Learning with Non-Cumulative Objective","date":"2023-07-11","arxiv_id":"2307.04957","repositories_listed":1,"syntology":null},{"url":"/paper/loss-dynamics-of-temporal-difference","slug":"loss-dynamics-of-temporal-difference","title":"Loss Dynamics of Temporal Difference Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04841","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-counterexample-guidance-for","slug":"probabilistic-counterexample-guidance-for","title":"Probabilistic Counterexample Guidance for Safer Reinforcement Learning (Extended Version)","date":"2023-07-10","arxiv_id":"2307.04927","repositories_listed":1,"syntology":null},{"url":"/paper/rltf-reinforcement-learning-from-unit-test","slug":"rltf-reinforcement-learning-from-unit-test","title":"RLTF: Reinforcement Learning from Unit Test Feedback","date":"2023-07-10","arxiv_id":"2307.04349","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rltf-reinforcement-learning-from-unit-test#ran","syntology_url":"https://syntology.ai/paper/2307.04349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04349"}},"official":{"repos":["zyq-scut/rltf"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/discovering-hierarchical-achievements-in-1","slug":"discovering-hierarchical-achievements-in-1","title":"Discovering Hierarchical Achievements in Reinforcement Learning via Contrastive Learning","date":"2023-07-07","arxiv_id":"2307.03486","repositories_listed":1,"syntology":null},{"url":"/paper/containergym-a-real-world-reinforcement","slug":"containergym-a-real-world-reinforcement","title":"ContainerGym: A Real-World Reinforcement Learning Benchmark for Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02991","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-observation-policies-in-observation","slug":"dynamic-observation-policies-in-observation","title":"Dynamic Observation Policies in Observation Cost-Sensitive Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02620","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dynamic-observation-policies-in-observation#ran","syntology_url":"https://syntology.ai/paper/2307.02620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02620"}},"official":{"repos":["cbellinger27/learning-when-to-observe-in-rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learning-symbolic-rules-over-abstract-meaning","slug":"learning-symbolic-rules-over-abstract-meaning","title":"Learning Symbolic Rules over Abstract Meaning Representations for Textual Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02689","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-symbolic-rules-over-abstract-meaning#ran","syntology_url":"https://syntology.ai/paper/2307.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02689"}},"official":{"repos":["ibm/loa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-objective-deep-reinforcement-learning-1","slug":"multi-objective-deep-reinforcement-learning-1","title":"Multi-objective Deep Reinforcement Learning for Mobile Edge Computing","date":"2023-07-05","arxiv_id":"2307.14346","repositories_listed":1,"syntology":null},{"url":"/paper/deep-attention-q-network-for-personalized","slug":"deep-attention-q-network-for-personalized","title":"Deep Attention Q-Network for Personalized Treatment Recommendation","date":"2023-07-04","arxiv_id":"2307.01519","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-model-equivalence-for-risk","slug":"distributional-model-equivalence-for-risk","title":"Distributional Model Equivalence for Risk-Sensitive Reinforcement Learning","date":"2023-07-04","arxiv_id":"2307.01708","repositories_listed":1,"syntology":null},{"url":"/paper/environmental-effects-on-emergent-strategy-in","slug":"environmental-effects-on-emergent-strategy-in","title":"Environmental effects on emergent strategy in micro-scale multi-agent reinforcement learning","date":"2023-07-03","arxiv_id":"2307.00994","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-reinforcement-learning-for-online","slug":"end-to-end-reinforcement-learning-for-online","title":"Learning Coverage Paths in Unknown Environments with Deep Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.16978","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-reinforcement-learning-for-online#ran","syntology_url":"https://syntology.ai/paper/2306.16978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16978"}},"official":{"repos":["arvijj/rl-cpp"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/safe-model-based-multi-agent-mean-field","slug":"safe-model-based-multi-agent-mean-field","title":"Safe Model-Based Multi-Agent Mean-Field Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.17052","repositories_listed":1,"syntology":null},{"url":"/paper/rl-3-boosting-meta-reinforcement-learning-via","slug":"rl-3-boosting-meta-reinforcement-learning-via","title":"RL$^3$: Boosting Meta Reinforcement Learning via RL inside RL$^2$","date":"2023-06-28","arxiv_id":"2306.15909","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-truss-design-with-reinforcement","slug":"automatic-truss-design-with-reinforcement","title":"Automatic Truss Design with Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15182","repositories_listed":1,"syntology":null},{"url":"/paper/creating-valid-adversarial-examples-of","slug":"creating-valid-adversarial-examples-of","title":"Creating Valid Adversarial Examples of Malware","date":"2023-06-23","arxiv_id":"2306.13587","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-ood-state-actions-supported-cross","slug":"beyond-ood-state-actions-supported-cross","title":"Beyond OOD State Actions: Supported Cross-Domain Offline Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.12755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-ood-state-actions-supported-cross#ran","syntology_url":"https://syntology.ai/paper/2306.12755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12755"}},"official":{"repos":["thuml/SPOT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-mixed-offline-reinforcement","slug":"harnessing-mixed-offline-reinforcement","title":"Harnessing Mixed Offline Reinforcement Learning Datasets via Trajectory Weighting","date":"2023-06-22","arxiv_id":"2306.13085","repositories_listed":1,"syntology":null},{"url":"/paper/robustneuralnetworks-jl-a-package-for-machine","slug":"robustneuralnetworks-jl-a-package-for-machine","title":"RobustNeuralNetworks.jl: a Package for Machine Learning and Data-Driven Control with Certified Robustness","date":"2023-06-22","arxiv_id":"2306.12612","repositories_listed":1,"syntology":null},{"url":"/paper/taco-temporal-latent-action-driven","slug":"taco-temporal-latent-action-driven","title":"TACO: Temporal Latent Action-Driven Contrastive Loss for Visual Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.13229","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/taco-temporal-latent-action-driven#ran","syntology_url":"https://syntology.ai/paper/2306.13229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13229"}},"official":{"repos":["frankzheng2022/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adcraft-an-advanced-reinforcement-learning","slug":"adcraft-an-advanced-reinforcement-learning","title":"AdCraft: An Advanced Reinforcement Learning Benchmark Environment for Search Engine Marketing Optimization","date":"2023-06-21","arxiv_id":"2306.11971","repositories_listed":1,"syntology":null},{"url":"/paper/state-wise-constrained-policy-optimization","slug":"state-wise-constrained-policy-optimization","title":"State-wise Constrained Policy Optimization","date":"2023-06-21","arxiv_id":"2306.12594","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-search-and-track-with-multiagent","slug":"adversarial-search-and-track-with-multiagent","title":"Adversarial Search and Tracking with Multiagent Reinforcement Learning in Sparsely Observable Environment","date":"2023-06-20","arxiv_id":"2306.11301","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generate-better-than-your-llm","slug":"learning-to-generate-better-than-your-llm","title":"Learning to Generate Better Than Your LLM","date":"2023-06-20","arxiv_id":"2306.11816","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-ordered-information-extraction-with","slug":"adaptive-ordered-information-extraction-with","title":"Adaptive Ordered Information Extraction with Deep Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10787","repositories_listed":1,"syntology":null},{"url":"/paper/cammarl-conformal-action-modeling-in-multi","slug":"cammarl-conformal-action-modeling-in-multi","title":"CAMMARL: Conformal Action Modeling in Multi Agent Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.11128","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/cammarl-conformal-action-modeling-in-multi#ran","syntology_url":"https://syntology.ai/paper/2306.11128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11128"}},"official":{"repos":["nikunj-gupta/conformal-agent-modelling"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-with-multitask","slug":"deep-reinforcement-learning-with-multitask","title":"Deep Reinforcement Learning with Task-Adaptive Retrieval via Hypernetwork","date":"2023-06-19","arxiv_id":"2306.10698","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-generalization-and-plasticity-for","slug":"enhancing-generalization-and-plasticity-for","title":"PLASTIC: Improving Input and Label Plasticity for Sample Efficient Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10711","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-quantum-variational-state","slug":"enhancing-quantum-variational-state","title":"Enhancing variational quantum state diagonalization using reinforcement learning techniques","date":"2023-06-19","arxiv_id":"2306.11086","repositories_listed":1,"syntology":null},{"url":"/paper/maximum-entropy-heterogeneous-agent-mirror","slug":"maximum-entropy-heterogeneous-agent-mirror","title":"Maximum Entropy Heterogeneous-Agent Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10715","repositories_listed":1,"syntology":null},{"url":"/paper/empowering-nlg-offline-reinforcement-learning-1","slug":"empowering-nlg-offline-reinforcement-learning-1","title":"Empowering NLG: Offline Reinforcement Learning for Informal Summarization in Online Domains","date":"2023-06-17","arxiv_id":"2306.17174","repositories_listed":1,"syntology":null},{"url":"/paper/genes-in-intelligent-agents","slug":"genes-in-intelligent-agents","title":"Genes in Intelligent Agents","date":"2023-06-17","arxiv_id":"2306.10225","repositories_listed":1,"syntology":null},{"url":"/paper/creating-multi-level-skill-hierarchies-in","slug":"creating-multi-level-skill-hierarchies-in","title":"Creating Multi-Level Skill Hierarchies in Reinforcement Learning","date":"2023-06-16","arxiv_id":"2306.09980","repositories_listed":1,"syntology":null},{"url":"/paper/jumanji-a-diverse-suite-of-scalable","slug":"jumanji-a-diverse-suite-of-scalable","title":"Jumanji: a Diverse Suite of Scalable Reinforcement Learning Environments in JAX","date":"2023-06-16","arxiv_id":"2306.09884","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jumanji-a-diverse-suite-of-scalable#ran","syntology_url":"https://syntology.ai/paper/2306.09884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09884"}},"official":{"repos":["instadeepai/jumanji"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semi-offline-reinforcement-learning-for","slug":"semi-offline-reinforcement-learning-for","title":"Semi-Offline Reinforcement Learning for Optimized Text Generation","date":"2023-06-16","arxiv_id":"2306.09712","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-learning-from-demonstration","slug":"a-framework-for-learning-from-demonstration","title":"A Framework for Learning from Demonstration with Minimal Human Effort","date":"2023-06-15","arxiv_id":"2306.09211","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-resource-scaling-of-5g-slices","slug":"generalizable-resource-scaling-of-5g-slices","title":"Generalizable Resource Scaling of 5G Slices using Constrained Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09290","repositories_listed":1,"syntology":null},{"url":"/paper/joint-path-planning-and-power-allocation-of-a","slug":"joint-path-planning-and-power-allocation-of-a","title":"Joint Path planning and Power Allocation of a Cellular-Connected UAV using Apprenticeship Learning via Deep Inverse Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.10071","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-memory-decision-transformer","slug":"recurrent-memory-decision-transformer","title":"Recurrent Action Transformer with Memory","date":"2023-06-15","arxiv_id":"2306.09459","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-memory-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2306.09459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09459"}},"official":{"repos":["airi-institute/rate"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/semantic-helm-a-human-readable-memory-for-1","slug":"semantic-helm-a-human-readable-memory-for-1","title":"Semantic HELM: A Human-Readable Memory for Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09312","repositories_listed":1,"syntology":null},{"url":"/paper/simplified-temporal-consistency-reinforcement","slug":"simplified-temporal-consistency-reinforcement","title":"Simplified Temporal Consistency Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09466","repositories_listed":1,"syntology":null},{"url":"/paper/curricular-subgoals-for-inverse-reinforcement","slug":"curricular-subgoals-for-inverse-reinforcement","title":"Curricular Subgoals for Inverse Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08232","repositories_listed":1,"syntology":null},{"url":"/paper/katakomba-tools-and-benchmarks-for-data","slug":"katakomba-tools-and-benchmarks-for-data","title":"Katakomba: Tools and Benchmarks for Data-Driven NetHack","date":"2023-06-14","arxiv_id":"2306.08772","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/katakomba-tools-and-benchmarks-for-data#ran","syntology_url":"https://syntology.ai/paper/2306.08772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08772"}},"official":{"repos":["corl-team/katakomba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mediated-multi-agent-reinforcement-learning","slug":"mediated-multi-agent-reinforcement-learning","title":"Mediated Multi-Agent Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08419","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mediated-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2306.08419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08419"}},"official":{"repos":["dimonenka/mediatedmarl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ocatari-object-centric-atari-2600","slug":"ocatari-object-centric-atari-2600","title":"OCAtari: Object-Centric Atari 2600 Reinforcement Learning Environments","date":"2023-06-14","arxiv_id":"2306.08649","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocatari-object-centric-atari-2600#ran","syntology_url":"https://syntology.ai/paper/2306.08649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08649"}},"official":{"repos":["k4ntz/oc_atari"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-versatile-multi-agent-reinforcement","slug":"a-versatile-multi-agent-reinforcement","title":"A Versatile Multi-Agent Reinforcement Learning Benchmark for Inventory Management","date":"2023-06-13","arxiv_id":"2306.07542","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-versatile-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.07542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07542"}},"official":{"repos":["victoryxl/replenishmentenv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-driven-linker-design","slug":"reinforcement-learning-driven-linker-design","title":"Reinforcement Learning-Driven Linker Design via Fast Attention-based Point Cloud Alignment","date":"2023-06-13","arxiv_id":"2306.08166","repositories_listed":1,"syntology":null},{"url":"/paper/generalizable-wireless-navigation-through","slug":"generalizable-wireless-navigation-through","title":"Digital Twin-Enhanced Wireless Indoor Navigation: Achieving Efficient Environment Sensing with Zero-Shot Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06766","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-wireless-navigation-through#ran","syntology_url":"https://syntology.ai/paper/2306.06766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06766"}},"official":{"repos":["panshark/pirl-win"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-efficacy-of-3d-point-cloud","slug":"on-the-efficacy-of-3d-point-cloud","title":"On the Efficacy of 3D Point Cloud Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06799","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-the-efficacy-of-3d-point-cloud#ran","syntology_url":"https://syntology.ai/paper/2306.06799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06799"}},"official":{"repos":["lz1oceani/pointcloud_rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-stacks-flexible-reinforcement","slug":"decision-stacks-flexible-reinforcement","title":"Decision Stacks: Flexible Reinforcement Learning via Modular Generative Models","date":"2023-06-09","arxiv_id":"2306.06253","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-reinforcement-learning-with","slug":"explaining-reinforcement-learning-with","title":"Explaining Reinforcement Learning with Shapley Values","date":"2023-06-09","arxiv_id":"2306.05810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explaining-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2306.05810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05810"}},"official":{"repos":["bath-reinforcement-learning-lab/sverl_icml_2023"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-semi-parametric-1","slug":"large-language-models-are-semi-parametric-1","title":"Large Language Models Are Semi-Parametric Reinforcement Learning Agents","date":"2023-06-09","arxiv_id":"2306.07929","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-semi-parametric-1#ran","syntology_url":"https://syntology.ai/paper/2306.07929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07929"}},"official":{"repos":["opendfm/rememberer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/treedqn-learning-to-minimize-branch-and-bound","slug":"treedqn-learning-to-minimize-branch-and-bound","title":"TreeDQN: Learning to minimize Branch-and-Bound tree","date":"2023-06-09","arxiv_id":"2306.05905","repositories_listed":1,"syntology":null},{"url":"/paper/learned-spatial-data-partitioning","slug":"learned-spatial-data-partitioning","title":"Learned spatial data partitioning","date":"2023-06-08","arxiv_id":"2306.04846","repositories_listed":1,"syntology":null},{"url":"/paper/progression-cognition-reinforcement-learning","slug":"progression-cognition-reinforcement-learning","title":"Progression Cognition Reinforcement Learning with Prioritized Experience for Multi-Vehicle Pursuit","date":"2023-06-08","arxiv_id":"2306.05016","repositories_listed":1,"syntology":null},{"url":"/paper/agent-performing-autonomous-stock-trading","slug":"agent-performing-autonomous-stock-trading","title":"Agent Performing Autonomous Stock Trading under Good and Bad Situations","date":"2023-06-06","arxiv_id":"2306.03985","repositories_listed":1,"syntology":null},{"url":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-interpretable-compressive-text","slug":"efficient-and-interpretable-compressive-text","title":"Efficient and Interpretable Compressive Text Summarisation with Unsupervised Dual-Agent Reinforcement Learning","date":"2023-06-06","arxiv_id":"2306.03415","repositories_listed":1,"syntology":null},{"url":"/paper/mildly-constrained-evaluation-policy-for","slug":"mildly-constrained-evaluation-policy-for","title":"Mildly Constrained Evaluation Policy for Offline Reinforcement Learning","date":"2023-06-06","arxiv_id":"2306.03680","repositories_listed":1,"syntology":null},{"url":"/paper/vid2act-activate-offline-videos-for-visual-rl","slug":"vid2act-activate-offline-videos-for-visual-rl","title":"Model-Based Reinforcement Learning with Multi-Task Offline Pretraining","date":"2023-06-06","arxiv_id":"2306.03360","repositories_listed":1,"syntology":null}],"record_sha256":"a33d7c1dc39f7012a0a0fc8446dd4da48b24baf8301a86a2e596b5cd8f125a54","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}