{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/17","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":152,"rows_per_page":100,"rows":[1601,1700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/16","next":"/task/reinforcement-learning-1/papers/18","papers":[{"url":"/paper/planning-the-path-with-reinforcement-learning","slug":"planning-the-path-with-reinforcement-learning","title":"Planning the path with Reinforcement Learning: Optimal Robot Motion Planning in RoboCup Small Size League Environments","date":"2024-04-23","arxiv_id":"2404.15410","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-adaptive-control","slug":"reinforcement-learning-with-adaptive-control","title":"Reinforcement Learning with Adaptive Regularization for Safe Control of Critical Systems","date":"2024-04-23","arxiv_id":"2404.15199","repositories_listed":1,"syntology":null},{"url":"/paper/multi-view-disentanglement-for-reinforcement","slug":"multi-view-disentanglement-for-reinforcement","title":"Multi-view Disentanglement for Reinforcement Learning with Multiple Cameras","date":"2024-04-22","arxiv_id":"2404.14064","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-view-disentanglement-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2404.14064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14064"}},"official":{"repos":["uoe-agents/mvd"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-fine-tuning-of-llms-should","slug":"preference-fine-tuning-of-llms-should","title":"Preference Fine-Tuning of LLMs Should Leverage Suboptimal, On-Policy Data","date":"2024-04-22","arxiv_id":"2404.14367","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/preference-fine-tuning-of-llms-should#ran","syntology_url":"https://syntology.ai/paper/2404.14367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14367"}},"official":{"repos":["Asap7772/understanding-rlhf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/trajdeleter-enabling-trajectory-forgetting-in","slug":"trajdeleter-enabling-trajectory-forgetting-in","title":"TrajDeleter: Enabling Trajectory Forgetting in Offline Reinforcement Learning Agents","date":"2024-04-18","arxiv_id":"2404.12530","repositories_listed":1,"syntology":null},{"url":"/paper/course-recommender-systems-need-to-consider","slug":"course-recommender-systems-need-to-consider","title":"Course Recommender Systems Need to Consider the Job Market","date":"2024-04-16","arxiv_id":"2404.10876","repositories_listed":1,"syntology":null},{"url":"/paper/sustainability-of-data-center-digital-twins","slug":"sustainability-of-data-center-digital-twins","title":"Sustainability of Data Center Digital Twins with Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10786","repositories_listed":1,"syntology":null},{"url":"/paper/what-hides-behind-unfairness-exploring","slug":"what-hides-behind-unfairness-exploring","title":"What Hides behind Unfairness? Exploring Dynamics Fairness in Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10942","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-hides-behind-unfairness-exploring#ran","syntology_url":"https://syntology.ai/paper/2404.10942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10942"}},"official":{"repos":["familyld/insightfair"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/inferring-behavior-specific-context-improves","slug":"inferring-behavior-specific-context-improves","title":"Inferring Behavior-Specific Context Improves Zero-Shot Generalization in Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.09521","repositories_listed":1,"syntology":null},{"url":"/paper/dataset-reset-policy-optimization-for-rlhf","slug":"dataset-reset-policy-optimization-for-rlhf","title":"Dataset Reset Policy Optimization for RLHF","date":"2024-04-12","arxiv_id":"2404.08495","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dataset-reset-policy-optimization-for-rlhf#ran","syntology_url":"https://syntology.ai/paper/2404.08495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08495"}},"official":{"repos":["cornell-rl/drpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/enhancing-autonomous-vehicle-training-with","slug":"enhancing-autonomous-vehicle-training-with","title":"Enhancing Autonomous Vehicle Training with Language Model Integration and Critical Scenario Generation","date":"2024-04-12","arxiv_id":"2404.08570","repositories_listed":1,"syntology":null},{"url":"/paper/generalized-population-based-training-for","slug":"generalized-population-based-training-for","title":"Generalized Population-Based Training for Hyperparameter Optimization in Reinforcement Learning","date":"2024-04-12","arxiv_id":"2404.08233","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-population-based-training-for#ran","syntology_url":"https://syntology.ai/paper/2404.08233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08233"}},"official":{"repos":["emi-group/gpbt-pl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wroom-an-autonomous-driving-approach-for-off","slug":"wroom-an-autonomous-driving-approach-for-off","title":"WROOM: An Autonomous Driving Approach for Off-Road Navigation","date":"2024-04-12","arxiv_id":"2404.08855","repositories_listed":1,"syntology":null},{"url":"/paper/how-consistent-are-clinicians-evaluating-the","slug":"how-consistent-are-clinicians-evaluating-the","title":"How Consistent are Clinicians? Evaluating the Predictability of Sepsis Disease Progression with Dynamics Models","date":"2024-04-10","arxiv_id":"2404.07148","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-consistent-are-clinicians-evaluating-the#ran","syntology_url":"https://syntology.ai/paper/2404.07148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07148"}},"official":{"repos":["cmudig/ai-clinician-mimiciv"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-out-of-distribution-detection-for","slug":"rethinking-out-of-distribution-detection-for","title":"Rethinking Out-of-Distribution Detection for Reinforcement Learning: Advancing Methods for Evaluation and Detection","date":"2024-04-10","arxiv_id":"2404.07099","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-out-of-distribution-detection-for#ran","syntology_url":"https://syntology.ai/paper/2404.07099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07099"}},"official":{"repos":["linasnas/dexter"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humanoid-gym-reinforcement-learning-for","slug":"humanoid-gym-reinforcement-learning-for","title":"Humanoid-Gym: Reinforcement Learning for Humanoid Robot with Zero-Shot Sim2Real Transfer","date":"2024-04-08","arxiv_id":"2404.05695","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humanoid-gym-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2404.05695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05695"}},"official":{"repos":["roboterax/humanoid-gym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compositional-conservatism-a-transductive","slug":"compositional-conservatism-a-transductive","title":"Compositional Conservatism: A Transductive Approach in Offline Reinforcement Learning","date":"2024-04-06","arxiv_id":"2404.04682","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/compositional-conservatism-a-transductive#ran","syntology_url":"https://syntology.ai/paper/2404.04682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04682"}},"official":{"repos":["runamu/compositional-conservatism"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-policy-distillation-of","slug":"continual-policy-distillation-of","title":"Continual Policy Distillation of Reinforcement Learning-based Controllers for Soft Robotic In-Hand Manipulation","date":"2024-04-05","arxiv_id":"2404.04219","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-recommendation-for-optimizing-both","slug":"sequential-recommendation-for-optimizing-both","title":"Sequential Recommendation for Optimizing Both Immediate Feedback and Long-term Retention","date":"2024-04-04","arxiv_id":"2404.03637","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequential-recommendation-for-optimizing-both#ran","syntology_url":"https://syntology.ai/paper/2404.03637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03637"}},"official":{"repos":["applied-machine-learning-lab/dt4ier"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/electric-vehicle-routing-problem-for","slug":"electric-vehicle-routing-problem-for","title":"Electric Vehicle Routing Problem for Emergency Power Supply: Towards Telecom Base Station Relief","date":"2024-04-03","arxiv_id":"2404.02448","repositories_listed":1,"syntology":null},{"url":"/paper/sliceit-a-dual-simulator-framework-for","slug":"sliceit-a-dual-simulator-framework-for","title":"SliceIt! -- A Dual Simulator Framework for Learning Robot Food Slicing","date":"2024-04-03","arxiv_id":"2404.02569","repositories_listed":1,"syntology":null},{"url":"/paper/entity-centric-reinforcement-learning-for","slug":"entity-centric-reinforcement-learning-for","title":"Entity-Centric Reinforcement Learning for Object Manipulation from Pixels","date":"2024-04-01","arxiv_id":"2404.01220","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/entity-centric-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2404.01220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01220"}},"official":{"repos":["danhrmti/ecrl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-new-agronomists-language-models-are","slug":"the-new-agronomists-language-models-are","title":"The New Agronomists: Language Models are Experts in Crop Management","date":"2024-03-28","arxiv_id":"2403.19839","repositories_listed":1,"syntology":null},{"url":"/paper/from-two-dimensional-to-three-dimensional-1","slug":"from-two-dimensional-to-three-dimensional-1","title":"From Two-Dimensional to Three-Dimensional Environment with Q-Learning: Modeling Autonomous Navigation with Reinforcement Learning and no Libraries","date":"2024-03-27","arxiv_id":"2403.18219","repositories_listed":1,"syntology":null},{"url":"/paper/learning-the-optimal-power-flow-environment","slug":"learning-the-optimal-power-flow-environment","title":"Learning the Optimal Power Flow: Environment Design Matters","date":"2024-03-26","arxiv_id":"2403.17831","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-receding-horizon","slug":"reinforcement-learning-based-receding-horizon","title":"Reinforcement Learning-based Receding Horizon Control using Adaptive Control Barrier Functions for Safety-Critical Systems","date":"2024-03-26","arxiv_id":"2403.17338","repositories_listed":1,"syntology":null},{"url":"/paper/rl-for-consistency-models-faster-reward","slug":"rl-for-consistency-models-faster-reward","title":"RL for Consistency Models: Faster Reward Guided Text-to-Image Generation","date":"2024-03-25","arxiv_id":"2404.03673","repositories_listed":1,"syntology":null},{"url":"/paper/policy-mirror-descent-with-lookahead","slug":"policy-mirror-descent-with-lookahead","title":"Policy Mirror Descent with Lookahead","date":"2024-03-21","arxiv_id":"2403.14156","repositories_listed":1,"syntology":null},{"url":"/paper/equivariant-ensembles-and-regularization-for","slug":"equivariant-ensembles-and-regularization-for","title":"Equivariant Ensembles and Regularization for Reinforcement Learning in Map-based Path Planning","date":"2024-03-19","arxiv_id":"2403.12856","repositories_listed":1,"syntology":null},{"url":"/paper/hydra-a-hyper-agent-for-dynamic-compositional","slug":"hydra-a-hyper-agent-for-dynamic-compositional","title":"HYDRA: A Hyper Agent for Dynamic Compositional Visual Reasoning","date":"2024-03-19","arxiv_id":"2403.12884","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hydra-a-hyper-agent-for-dynamic-compositional#ran","syntology_url":"https://syntology.ai/paper/2403.12884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12884"}},"official":{"repos":["ControlNet/HYDRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-bifurcation-in-safe-reinforcement","slug":"policy-bifurcation-in-safe-reinforcement","title":"Policy Bifurcation in Safe Reinforcement Learning","date":"2024-03-19","arxiv_id":"2403.12847","repositories_listed":1,"syntology":null},{"url":"/paper/decomposing-control-lyapunov-functions-for","slug":"decomposing-control-lyapunov-functions-for","title":"Decomposing Control Lyapunov Functions for Efficient Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.12210","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-transformer-based-hyper-parameter","slug":"efficient-transformer-based-hyper-parameter","title":"Efficient Transformer-based Hyper-parameter Optimization for Resource-constrained IoT Environments","date":"2024-03-18","arxiv_id":"2403.12237","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-token-level","slug":"reinforcement-learning-with-token-level","title":"Reinforcement Learning with Token-level Feedback for Controllable Text Generation","date":"2024-03-18","arxiv_id":"2403.11558","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-reinforcement-learning-hierarchical","slug":"diffusion-reinforcement-learning-hierarchical","title":"Diffusion-Reinforcement Learning Hierarchical Motion Planning in Multi-agent Adversarial Games","date":"2024-03-16","arxiv_id":"2403.10794","repositories_listed":1,"syntology":null},{"url":"/paper/explorer-exploration-guided-reasoning-for","slug":"explorer-exploration-guided-reasoning-for","title":"EXPLORER: Exploration-guided Reasoning for Textual Reinforcement Learning","date":"2024-03-15","arxiv_id":"2403.10692","repositories_listed":1,"syntology":null},{"url":"/paper/easy-to-hard-generalization-scalable","slug":"easy-to-hard-generalization-scalable","title":"Easy-to-Hard Generalization: Scalable Alignment Beyond Human Supervision","date":"2024-03-14","arxiv_id":"2403.09472","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/easy-to-hard-generalization-scalable#ran","syntology_url":"https://syntology.ai/paper/2403.09472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09472"}},"official":{"repos":["edward-sun/easy-to-hard"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-describe-for-predicting-zero-shot","slug":"learning-to-describe-for-predicting-zero-shot","title":"Learning to Describe for Predicting Zero-shot Drug-Drug Interactions","date":"2024-03-13","arxiv_id":"2403.08377","repositories_listed":1,"syntology":null},{"url":"/paper/llm-assisted-light-leveraging-large-language","slug":"llm-assisted-light-leveraging-large-language","title":"LLM-Assisted Light: Leveraging Large Language Model Capabilities for Human-Mimetic Traffic Signal Control in Complex Urban Environments","date":"2024-03-13","arxiv_id":"2403.08337","repositories_listed":1,"syntology":null},{"url":"/paper/teams-rl-teaching-llms-to-teach-themselves","slug":"teams-rl-teaching-llms-to-teach-themselves","title":"TeaMs-RL: Teaching LLMs to Generate Better Instruction Datasets via Reinforcement Learning","date":"2024-03-13","arxiv_id":"2403.08694","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-gain-scheduling-using-reinforcement","slug":"adaptive-gain-scheduling-using-reinforcement","title":"Adaptive Gain Scheduling using Reinforcement Learning for Quadcopter Control","date":"2024-03-12","arxiv_id":"2403.07216","repositories_listed":1,"syntology":null},{"url":"/paper/advantage-aware-policy-optimization-for","slug":"advantage-aware-policy-optimization-for","title":"A2PO: Towards Effective Offline Reinforcement Learning from an Advantage-aware Perspective","date":"2024-03-12","arxiv_id":"2403.07262","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-cross-modal-transfer-of","slug":"zero-shot-cross-modal-transfer-of","title":"Zero-shot cross-modal transfer of Reinforcement Learning policies through a Global Workspace","date":"2024-03-07","arxiv_id":"2403.04588","repositories_listed":1,"syntology":null},{"url":"/paper/belief-enriched-pessimistic-q-learning","slug":"belief-enriched-pessimistic-q-learning","title":"Belief-Enriched Pessimistic Q-Learning against Adversarial State Perturbations","date":"2024-03-06","arxiv_id":"2403.04050","repositories_listed":1,"syntology":{"n":24,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":24,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/belief-enriched-pessimistic-q-learning#ran","syntology_url":"https://syntology.ai/paper/2403.04050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04050"}},"official":{"repos":["sliencerx/belief-enriched-robust-q-learning"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/sampling-based-safe-reinforcement-learning","slug":"sampling-based-safe-reinforcement-learning","title":"Sampling-based Safe Reinforcement Learning for Nonlinear Dynamical Systems","date":"2024-03-06","arxiv_id":"2403.04007","repositories_listed":1,"syntology":null},{"url":"/paper/splagger-split-aggregation-for-meta","slug":"splagger-split-aggregation-for-meta","title":"SplAgger: Split Aggregation for Meta-Reinforcement Learning","date":"2024-03-05","arxiv_id":"2403.03020","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/splagger-split-aggregation-for-meta#ran","syntology_url":"https://syntology.ai/paper/2403.03020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03020"}},"official":{"repos":["jacooba/hyper"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-validity-of-automatically","slug":"improving-the-validity-of-automatically","title":"Improving the Validity of Automatically Generated Feedback via Reinforcement Learning","date":"2024-03-02","arxiv_id":"2403.01304","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-the-validity-of-automatically#ran","syntology_url":"https://syntology.ai/paper/2403.01304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01304"}},"official":{"repos":["umass-ml4ed/feedback-gen-dpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/efficientzero-v2-mastering-discrete-and","slug":"efficientzero-v2-mastering-discrete-and","title":"EfficientZero V2: Mastering Discrete and Continuous Control with Limited Data","date":"2024-03-01","arxiv_id":"2403.00564","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficientzero-v2-mastering-discrete-and#ran","syntology_url":"https://syntology.ai/paper/2403.00564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00564"}},"official":{"repos":["shengjiewang-jason/efficientzerov2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/curiosity-driven-red-teaming-for-large","slug":"curiosity-driven-red-teaming-for-large","title":"Curiosity-driven Red-teaming for Large Language Models","date":"2024-02-29","arxiv_id":"2402.19464","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curiosity-driven-red-teaming-for-large#ran","syntology_url":"https://syntology.ai/paper/2402.19464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19464"}},"official":{"repos":["improbable-ai/curiosity_redteam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-long-term-recommendation-with-bi","slug":"enhancing-long-term-recommendation-with-bi","title":"Large Language Models are Learnable Planners for Long-Term Recommendation","date":"2024-02-29","arxiv_id":"2403.00843","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-long-term-recommendation-with-bi#ran","syntology_url":"https://syntology.ai/paper/2403.00843","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00843"}},"official":{"repos":["jizhi-zhang/billp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rebandit-random-effects-based-online-rl","slug":"rebandit-random-effects-based-online-rl","title":"reBandit: Random Effects based Online RL algorithm for Reducing Cannabis Use","date":"2024-02-27","arxiv_id":"2402.17739","repositories_listed":1,"syntology":null},{"url":"/paper/craftax-a-lightning-fast-benchmark-for-open","slug":"craftax-a-lightning-fast-benchmark-for-open","title":"Craftax: A Lightning-Fast Benchmark for Open-Ended Reinforcement Learning","date":"2024-02-26","arxiv_id":"2402.16801","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":14,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/craftax-a-lightning-fast-benchmark-for-open#ran","syntology_url":"https://syntology.ai/paper/2402.16801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16801"}},"official":{"repos":["michaeltmatthews/craftax"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/feedback-efficient-online-fine-tuning-of","slug":"feedback-efficient-online-fine-tuning-of","title":"Feedback Efficient Online Fine-Tuning of Diffusion Models","date":"2024-02-26","arxiv_id":"2402.16359","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/feedback-efficient-online-fine-tuning-of#ran","syntology_url":"https://syntology.ai/paper/2402.16359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16359"}},"official":null}},{"url":"/paper/flexible-robust-beamforming-for-multibeam","slug":"flexible-robust-beamforming-for-multibeam","title":"Flexible Robust Beamforming for Multibeam Satellite Downlink using Reinforcement Learning","date":"2024-02-26","arxiv_id":"2402.16563","repositories_listed":1,"syntology":null},{"url":"/paper/gennbv-generalizable-next-best-view-policy","slug":"gennbv-generalizable-next-best-view-policy","title":"GenNBV: Generalizable Next-Best-View Policy for Active 3D Reconstruction","date":"2024-02-25","arxiv_id":"2402.16174","repositories_listed":1,"syntology":null},{"url":"/paper/how-can-llm-guide-rl-a-value-based-approach","slug":"how-can-llm-guide-rl-a-value-based-approach","title":"How Can LLM Guide RL? A Value-Based Approach","date":"2024-02-25","arxiv_id":"2402.16181","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/how-can-llm-guide-rl-a-value-based-approach#ran","syntology_url":"https://syntology.ai/paper/2402.16181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16181"}},"official":{"repos":["agentification/language-integrated-vi"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distributionally-robust-off-dynamics","slug":"distributionally-robust-off-dynamics","title":"Distributionally Robust Off-Dynamics Reinforcement Learning: Provable Efficiency with Linear Function Approximation","date":"2024-02-23","arxiv_id":"2402.15399","repositories_listed":1,"syntology":null},{"url":"/paper/easyrl4rec-a-user-friendly-code-library-for","slug":"easyrl4rec-a-user-friendly-code-library-for","title":"EasyRL4Rec: An Easy-to-use Library for Reinforcement Learning Based Recommender Systems","date":"2024-02-23","arxiv_id":"2402.15164","repositories_listed":1,"syntology":null},{"url":"/paper/himap-learning-heuristics-informed-policies","slug":"himap-learning-heuristics-informed-policies","title":"HiMAP: Learning Heuristics-Informed Policies for Large-Scale Multi-Agent Pathfinding","date":"2024-02-23","arxiv_id":"2402.15546","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/himap-learning-heuristics-informed-policies#ran","syntology_url":"https://syntology.ai/paper/2402.15546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15546"}},"official":{"repos":["kaist-silab/himap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distinctive-image-captioning-leveraging","slug":"distinctive-image-captioning-leveraging","title":"Distinctive Image Captioning: Leveraging Ground Truth Captions in CLIP Guided Reinforcement Learning","date":"2024-02-21","arxiv_id":"2402.13936","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-dynamic-multi","slug":"reinforcement-learning-with-dynamic-multi","title":"Dynamic Multi-Reward Weighting for Multi-Style Controllable Generation","date":"2024-02-21","arxiv_id":"2402.14146","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-worst-case-attacks-robust-rl-with","slug":"beyond-worst-case-attacks-robust-rl-with","title":"Beyond Worst-case Attacks: Robust RL with Adaptive Defense via Non-dominated Policies","date":"2024-02-20","arxiv_id":"2402.12673","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/beyond-worst-case-attacks-robust-rl-with#ran","syntology_url":"https://syntology.ai/paper/2402.12673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12673"}},"official":{"repos":["umd-huang-lab/protected"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/more-3s-multimodal-based-offline","slug":"more-3s-multimodal-based-offline","title":"MORE-3S:Multimodal-based Offline Reinforcement Learning with Shared Semantic Spaces","date":"2024-02-20","arxiv_id":"2402.12845","repositories_listed":1,"syntology":null},{"url":"/paper/reflect-rl-two-player-online-rl-fine-tuning","slug":"reflect-rl-two-player-online-rl-fine-tuning","title":"Reflect-RL: Two-Player Online RL Fine-Tuning for LMs","date":"2024-02-20","arxiv_id":"2402.12621","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reflect-rl-two-player-online-rl-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2402.12621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12621"}},"official":{"repos":["zhourunlong/reflect-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/xrl-bench-a-benchmark-for-evaluating-and","slug":"xrl-bench-a-benchmark-for-evaluating-and","title":"XRL-Bench: A Benchmark for Evaluating and Comparing Explainable Reinforcement Learning Techniques","date":"2024-02-20","arxiv_id":"2402.12685","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-evaluation-of-ai-feedback-for","slug":"a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-critical-evaluation-of-ai-feedback-for#ran","syntology_url":"https://syntology.ai/paper/2402.12366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12366"}},"official":{"repos":["architsharma97/dpo-rlaif"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/modelling-crypto-markets-by-multi-agent","slug":"modelling-crypto-markets-by-multi-agent","title":"Modelling crypto markets by multi-agent reinforcement learning","date":"2024-02-16","arxiv_id":"2402.10803","repositories_listed":1,"syntology":null},{"url":"/paper/policy-learning-for-off-dynamics-rl-with","slug":"policy-learning-for-off-dynamics-rl-with","title":"Policy Learning for Off-Dynamics RL with Deficient Support","date":"2024-02-16","arxiv_id":"2402.10765","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-learning-for-off-dynamics-rl-with#ran","syntology_url":"https://syntology.ai/paper/2402.10765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10765"}},"official":{"repos":["linhlpv/DADS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/jack-of-all-trades-master-of-some-a-multi","slug":"jack-of-all-trades-master-of-some-a-multi","title":"Jack of All Trades, Master of Some, a Multi-Purpose Transformer Agent","date":"2024-02-15","arxiv_id":"2402.09844","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/jack-of-all-trades-master-of-some-a-multi#ran","syntology_url":"https://syntology.ai/paper/2402.09844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09844"}},"official":{"repos":["huggingface/jat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-and-risk-aware-offline-multi","slug":"conservative-and-risk-aware-offline-multi","title":"Conservative and Risk-Aware Offline Multi-Agent Reinforcement Learning","date":"2024-02-13","arxiv_id":"2402.08421","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conservative-and-risk-aware-offline-multi#ran","syntology_url":"https://syntology.ai/paper/2402.08421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08421"}},"official":{"repos":["eslam211/conservative-and-distributional-marl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-inverse-reinforcement-learning","slug":"hybrid-inverse-reinforcement-learning","title":"Hybrid Inverse Reinforcement Learning","date":"2024-02-13","arxiv_id":"2402.08848","repositories_listed":1,"syntology":null},{"url":"/paper/deceptive-path-planning-via-reinforcement","slug":"deceptive-path-planning-via-reinforcement","title":"Deceptive Path Planning via Reinforcement Learning with Graph Neural Networks","date":"2024-02-09","arxiv_id":"2402.06552","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-regularized-token-level-policy","slug":"entropy-regularized-token-level-policy","title":"Entropy-Regularized Token-Level Policy Optimization for Language Agent Reinforcement","date":"2024-02-09","arxiv_id":"2402.06700","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/entropy-regularized-token-level-policy#ran","syntology_url":"https://syntology.ai/paper/2402.06700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06700"}},"official":{"repos":["morning9393/etpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/monitored-markov-decision-processes","slug":"monitored-markov-decision-processes","title":"Monitored Markov Decision Processes","date":"2024-02-09","arxiv_id":"2402.06819","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monitored-markov-decision-processes#ran","syntology_url":"https://syntology.ai/paper/2402.06819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06819"}},"official":{"repos":["amiithinks/mon_mdp_aamas24"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-rl-for-mean-field-games-is-not","slug":"model-based-rl-for-mean-field-games-is-not","title":"Model-Based RL for Mean-Field Games is not Statistically Harder than Single-Agent RL","date":"2024-02-08","arxiv_id":"2402.05724","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-rl-for-mean-field-games-is-not#ran","syntology_url":"https://syntology.ai/paper/2402.05724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05724"}},"official":{"repos":["jiaweihhuang/heuristic_mebp"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-timescale-ensemble-q-learning-for","slug":"multi-timescale-ensemble-q-learning-for","title":"Multi-Timescale Ensemble Q-learning for Markov Decision Process Policy Optimization","date":"2024-02-08","arxiv_id":"2402.05476","repositories_listed":1,"syntology":null},{"url":"/paper/training-large-language-models-for-reasoning","slug":"training-large-language-models-for-reasoning","title":"Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.05808","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":19,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/training-large-language-models-for-reasoning#ran","syntology_url":"https://syntology.ai/paper/2402.05808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05808"}},"official":{"repos":["woooodyy/llm-reverse-curriculum-rl"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/oil-ad-an-anomaly-detection-framework-for","slug":"oil-ad-an-anomaly-detection-framework-for","title":"OIL-AD: An Anomaly Detection Framework for Sequential Decision Sequences","date":"2024-02-07","arxiv_id":"2402.04567","repositories_listed":1,"syntology":null},{"url":"/paper/qgfn-controllable-greediness-with-action","slug":"qgfn-controllable-greediness-with-action","title":"QGFN: Controllable Greediness with Action Values","date":"2024-02-07","arxiv_id":"2402.05234","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/qgfn-controllable-greediness-with-action#ran","syntology_url":"https://syntology.ai/paper/2402.05234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05234"}},"official":{"repos":["yunglau/QGFN"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safety-filters-for-black-box-dynamical","slug":"safety-filters-for-black-box-dynamical","title":"Safety Filters for Black-Box Dynamical Systems by Learning Discriminating Hyperplanes","date":"2024-02-07","arxiv_id":"2402.05279","repositories_listed":1,"syntology":null},{"url":"/paper/entropy-regularized-diffusion-policy-with-q","slug":"entropy-regularized-diffusion-policy-with-q","title":"Entropy-regularized Diffusion Policy with Q-Ensembles for Offline Reinforcement Learning","date":"2024-02-06","arxiv_id":"2402.04080","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entropy-regularized-diffusion-policy-with-q#ran","syntology_url":"https://syntology.ai/paper/2402.04080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04080"}},"official":{"repos":["ruoqizzz/entropy-offlineRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/logical-specifications-guided-dynamic-task","slug":"logical-specifications-guided-dynamic-task","title":"Logical Specifications-guided Dynamic Task Sampling for Reinforcement Learning Agents","date":"2024-02-06","arxiv_id":"2402.03678","repositories_listed":1,"syntology":null},{"url":"/paper/rl-vlm-f-reinforcement-learning-from-vision","slug":"rl-vlm-f-reinforcement-learning-from-vision","title":"RL-VLM-F: Reinforcement Learning from Vision Language Foundation Model Feedback","date":"2024-02-06","arxiv_id":"2402.03681","repositories_listed":1,"syntology":null},{"url":"/paper/seabo-a-simple-search-based-method-for","slug":"seabo-a-simple-search-based-method-for","title":"SEABO: A Simple Search-Based Method for Offline Imitation Learning","date":"2024-02-06","arxiv_id":"2402.03807","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seabo-a-simple-search-based-method-for#ran","syntology_url":"https://syntology.ai/paper/2402.03807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03807"}},"official":{"repos":["dmksjfl/seabo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-long-delayed-reinforcement-learning","slug":"boosting-long-delayed-reinforcement-learning","title":"Boosting Reinforcement Learning with Strongly Delayed Feedback Through Auxiliary Short Delays","date":"2024-02-05","arxiv_id":"2402.03141","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/boosting-long-delayed-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2402.03141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03141"}},"official":{"repos":["qingyuanwunothing/ad-rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-reinforcement-learning-models-is","slug":"fine-tuning-reinforcement-learning-models-is","title":"Fine-tuning Reinforcement Learning Models is Secretly a Forgetting Mitigation Problem","date":"2024-02-05","arxiv_id":"2402.02868","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-teaching-regularization","slug":"learning-from-teaching-regularization","title":"Learning from Teaching Regularization: Generalizable Correlations Should be Easy to Imitate","date":"2024-02-05","arxiv_id":"2402.02769","repositories_listed":1,"syntology":null},{"url":"/paper/open-rl-benchmark-comprehensive-tracked","slug":"open-rl-benchmark-comprehensive-tracked","title":"Open RL Benchmark: Comprehensive Tracked Experiments for Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.03046","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-rl-benchmark-comprehensive-tracked#ran","syntology_url":"https://syntology.ai/paper/2402.03046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03046"}},"official":null}},{"url":"/paper/replication-of-impedance-identification","slug":"replication-of-impedance-identification","title":"Replication of Impedance Identification Experiments on a Reinforcement-Learning-Controlled Digital Twin of Human Elbows","date":"2024-02-05","arxiv_id":"2402.02904","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-the-role-of-proxy-rewards-in","slug":"rethinking-the-role-of-proxy-rewards-in","title":"Rethinking the Role of Proxy Rewards in Language Model Alignment","date":"2024-02-02","arxiv_id":"2402.03469","repositories_listed":1,"syntology":null},{"url":"/paper/stepcoder-improve-code-generation-with","slug":"stepcoder-improve-code-generation-with","title":"StepCoder: Improve Code Generation with Reinforcement Learning from Compiler Feedback","date":"2024-02-02","arxiv_id":"2402.01391","repositories_listed":1,"syntology":null},{"url":"/paper/to-the-max-reinventing-reward-in","slug":"to-the-max-reinventing-reward-in","title":"To the Max: Reinventing Reward in Reinforcement Learning","date":"2024-02-02","arxiv_id":"2402.01361","repositories_listed":1,"syntology":null},{"url":"/paper/developing-a-multi-agent-and-self-adaptive","slug":"developing-a-multi-agent-and-self-adaptive","title":"Developing A Multi-Agent and Self-Adaptive Framework with Deep Reinforcement Learning for Dynamic Portfolio Risk Management","date":"2024-02-01","arxiv_id":"2402.00515","repositories_listed":1,"syntology":null},{"url":"/paper/expert-proximity-as-surrogate-rewards-for","slug":"expert-proximity-as-surrogate-rewards-for","title":"Expert Proximity as Surrogate Rewards for Single Demonstration Imitation Learning","date":"2024-02-01","arxiv_id":"2402.01057","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/expert-proximity-as-surrogate-rewards-for#ran","syntology_url":"https://syntology.ai/paper/2402.01057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01057"}},"official":{"repos":["stanl1y/tdil"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-approximate-model-based-shielding","slug":"leveraging-approximate-model-based-shielding","title":"Leveraging Approximate Model-based Shielding for Probabilistic Safety Guarantees in Continuous Environments","date":"2024-02-01","arxiv_id":"2402.00816","repositories_listed":1,"syntology":null},{"url":"/paper/odice-revealing-the-mystery-of-distribution","slug":"odice-revealing-the-mystery-of-distribution","title":"ODICE: Revealing the Mystery of Distribution Correction Estimation via Orthogonal-gradient Update","date":"2024-02-01","arxiv_id":"2402.00348","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":1,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/odice-revealing-the-mystery-of-distribution#ran","syntology_url":"https://syntology.ai/paper/2402.00348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00348"}},"official":{"repos":["maoliyuan/odice-pytorch"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-policy-gradient-primal-dual-algorithm-for","slug":"a-policy-gradient-primal-dual-algorithm-for","title":"A Policy Gradient Primal-Dual Algorithm for Constrained MDPs with Uniform PAC Guarantees","date":"2024-01-31","arxiv_id":"2401.17780","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-replay-in-world-models-for","slug":"augmenting-replay-in-world-models-for","title":"Augmenting Replay in World Models for Continual Reinforcement Learning","date":"2024-01-30","arxiv_id":"2401.16650","repositories_listed":1,"syntology":null},{"url":"/paper/m2curl-sample-efficient-multimodal","slug":"m2curl-sample-efficient-multimodal","title":"M2CURL: Sample-Efficient Multimodal Reinforcement Learning via Self-Supervised Representation Learning for Robotic Manipulation","date":"2024-01-30","arxiv_id":"2401.17032","repositories_listed":1,"syntology":null},{"url":"/paper/a-centralized-reinforcement-learning","slug":"a-centralized-reinforcement-learning","title":"LEACH-RLC: Enhancing IoT Data Transmission with Optimized Clustering and Reinforcement Learning","date":"2024-01-28","arxiv_id":"2401.15767","repositories_listed":1,"syntology":null}],"record_sha256":"acce9e379b0453e79a3c6257d4bf080016a1c67fb85126e4e1c7b33d39181b8d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}