{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/13","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":152,"rows_per_page":100,"rows":[1201,1300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/12","next":"/task/reinforcement-learning-1/papers/14","papers":[{"url":"/paper/openvlthinker-an-early-exploration-to-complex","slug":"openvlthinker-an-early-exploration-to-complex","title":"OpenVLThinker: An Early Exploration to Complex Vision-Language Reasoning via Iterative Self-Improvement","date":"2025-03-21","arxiv_id":"2503.17352","repositories_listed":1,"syntology":{"n":21,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":12,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/openvlthinker-an-early-exploration-to-complex#ran","syntology_url":"https://syntology.ai/paper/2503.17352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.17352"}},"official":{"repos":["yihedeng9/openvlthinker"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/cls-rl-image-classification-with-rule-based","slug":"cls-rl-image-classification-with-rule-based","title":"Think or Not Think: A Study of Explicit Thinking in Rule-Based Visual Reinforcement Fine-Tuning","date":"2025-03-20","arxiv_id":"2503.16188","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cls-rl-image-classification-with-rule-based#ran","syntology_url":"https://syntology.ai/paper/2503.16188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16188"}},"official":{"repos":["minglllli/CLS-RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fin-r1-a-large-language-model-for-financial","slug":"fin-r1-a-large-language-model-for-financial","title":"Fin-R1: A Large Language Model for Financial Reasoning through Reinforcement Learning","date":"2025-03-20","arxiv_id":"2503.16252","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-heuristics-to","slug":"reinforcement-learning-based-heuristics-to","title":"Reinforcement Learning-based Heuristics to Guide Domain-Independent Dynamic Programming","date":"2025-03-20","arxiv_id":"2503.16371","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-reasoning-in-small","slug":"reinforcement-learning-for-reasoning-in-small","title":"Reinforcement Learning for Reasoning in Small LLMs: What Works and What Doesn't","date":"2025-03-20","arxiv_id":"2503.16219","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-for-reasoning-in-small#ran","syntology_url":"https://syntology.ai/paper/2503.16219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16219"}},"official":{"repos":["knoveleng/open-rs"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-overthinking-a-survey-on-efficient","slug":"stop-overthinking-a-survey-on-efficient","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","date":"2025-03-20","arxiv_id":"2503.16419","repositories_listed":1,"syntology":null},{"url":"/paper/neural-lyapunov-function-approximation-with","slug":"neural-lyapunov-function-approximation-with","title":"Neural Lyapunov Function Approximation with Self-Supervised Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.15629","repositories_listed":1,"syntology":null},{"url":"/paper/cosmos-reason1-from-physical-common-sense-to","slug":"cosmos-reason1-from-physical-common-sense-to","title":"Cosmos-Reason1: From Physical Common Sense To Embodied Reasoning","date":"2025-03-18","arxiv_id":"2503.15558","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cosmos-reason1-from-physical-common-sense-to#ran","syntology_url":"https://syntology.ai/paper/2503.15558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15558"}},"official":{"repos":["nvidia-cosmos/cosmos-reason1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/med-r1-reinforcement-learning-for","slug":"med-r1-reinforcement-learning-for","title":"Med-R1: Reinforcement Learning for Generalizable Medical Reasoning in Vision-Language Models","date":"2025-03-18","arxiv_id":"2503.13939","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-motion-imitation","slug":"reinforcement-learning-based-motion-imitation","title":"Reinforcement learning-based motion imitation for physiologically plausible musculoskeletal motor control","date":"2025-03-18","arxiv_id":"2503.14637","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-llm-reasoning-with-iterative-dpo-a","slug":"enhancing-llm-reasoning-with-iterative-dpo-a","title":"Enhancing LLM Reasoning with Iterative DPO: A Comprehensive Empirical Investigation","date":"2025-03-17","arxiv_id":"2503.12854","repositories_listed":1,"syntology":null},{"url":"/paper/terl-large-scale-multi-target-encirclement","slug":"terl-large-scale-multi-target-encirclement","title":"TERL: Large-Scale Multi-Target Encirclement Using Transformer-Enhanced Reinforcement Learning","date":"2025-03-16","arxiv_id":"2503.12395","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reset-in-target-search-problems","slug":"learning-to-reset-in-target-search-problems","title":"Learning to reset in target search problems","date":"2025-03-14","arxiv_id":"2503.11330","repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-alignment-training-diffusion","slug":"towards-better-alignment-training-diffusion","title":"Towards Better Alignment: Training Diffusion Models with Reinforcement Learning Against Sparse Rewards","date":"2025-03-14","arxiv_id":"2503.11240","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-better-alignment-training-diffusion#ran","syntology_url":"https://syntology.ai/paper/2503.11240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.11240"}},"official":{"repos":["hu-zijing/b2-diffurl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-evaluation-of-online-moderation","slug":"scalable-evaluation-of-online-moderation","title":"Scalable Evaluation of Online Facilitation Strategies via Synthetic Simulation of Discussions","date":"2025-03-13","arxiv_id":"2503.16505","repositories_listed":1,"syntology":null},{"url":"/paper/regulatory-dna-sequence-design-with","slug":"regulatory-dna-sequence-design-with","title":"Regulatory DNA sequence Design with Reinforcement Learning","date":"2025-03-11","arxiv_id":"2503.07981","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regulatory-dna-sequence-design-with#ran","syntology_url":"https://syntology.ai/paper/2503.07981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07981"}},"official":{"repos":["yangzhao1230/taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-max-making-rl-practical-for-autonomous","slug":"v-max-making-rl-practical-for-autonomous","title":"V-Max: A Reinforcement Learning Framework for Autonomous Driving","date":"2025-03-11","arxiv_id":"2503.08388","repositories_listed":1,"syntology":null},{"url":"/paper/alphadrive-unleashing-the-power-of-vlms-in","slug":"alphadrive-unleashing-the-power-of-vlms-in","title":"AlphaDrive: Unleashing the Power of VLMs in Autonomous Driving via Reinforcement Learning and Reasoning","date":"2025-03-10","arxiv_id":"2503.07608","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alphadrive-unleashing-the-power-of-vlms-in#ran","syntology_url":"https://syntology.ai/paper/2503.07608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07608"}},"official":{"repos":["hustvl/alphadrive"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lmm-r1-empowering-3b-lmms-with-strong","slug":"lmm-r1-empowering-3b-lmms-with-strong","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","date":"2025-03-10","arxiv_id":"2503.07536","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lmm-r1-empowering-3b-lmms-with-strong#ran","syntology_url":"https://syntology.ai/paper/2503.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07536"}},"official":null}},{"url":"/paper/mm-eureka-exploring-visual-aha-moment-with","slug":"mm-eureka-exploring-visual-aha-moment-with","title":"MM-Eureka: Exploring Visual Aha Moment with Rule-based Large-scale Reinforcement Learning","date":"2025-03-10","arxiv_id":"2503.07365","repositories_listed":1,"syntology":null},{"url":"/paper/visrl-intention-driven-visual-perception-via","slug":"visrl-intention-driven-visual-perception-via","title":"VisRL: Intention-Driven Visual Perception via Reinforced Reasoning","date":"2025-03-10","arxiv_id":"2503.07523","repositories_listed":1,"syntology":null},{"url":"/paper/agent-models-internalizing-chain-of-action","slug":"agent-models-internalizing-chain-of-action","title":"Agent models: Internalizing Chain-of-Action Generation into Reasoning models","date":"2025-03-09","arxiv_id":"2503.06580","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agent-models-internalizing-chain-of-action#ran","syntology_url":"https://syntology.ai/paper/2503.06580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06580"}},"official":{"repos":["adam-bjtu/autocoa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-proof-of-polynomial-inequalities","slug":"automated-proof-of-polynomial-inequalities","title":"Automated Proof of Polynomial Inequalities via Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.06592","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automated-proof-of-polynomial-inequalities#ran","syntology_url":"https://syntology.ai/paper/2503.06592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06592"}},"official":{"repos":["blliu6/appirl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/swift-hydra-self-reinforcing-generative","slug":"swift-hydra-self-reinforcing-generative","title":"Swift Hydra: Self-Reinforcing Generative Framework for Anomaly Detection with Multiple Mamba Models","date":"2025-03-09","arxiv_id":"2503.06413","repositories_listed":1,"syntology":null},{"url":"/paper/vision-r1-incentivizing-reasoning-capability","slug":"vision-r1-incentivizing-reasoning-capability","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","date":"2025-03-09","arxiv_id":"2503.06749","repositories_listed":1,"syntology":null},{"url":"/paper/policy-constraint-by-only-support-constraint","slug":"policy-constraint-by-only-support-constraint","title":"Policy Constraint by Only Support Constraint for Offline Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05207","repositories_listed":1,"syntology":null},{"url":"/paper/lessons-learned-from-field-demonstrations-of","slug":"lessons-learned-from-field-demonstrations-of","title":"Lessons learned from field demonstrations of model predictive control and reinforcement learning for residential and commercial HVAC: A review","date":"2025-03-06","arxiv_id":"2503.05022","repositories_listed":1,"syntology":null},{"url":"/paper/cognitive-behaviors-that-enable-self","slug":"cognitive-behaviors-that-enable-self","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","date":"2025-03-03","arxiv_id":"2503.01307","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cognitive-behaviors-that-enable-self#ran","syntology_url":"https://syntology.ai/paper/2503.01307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01307"}},"official":{"repos":["kanishkg/cognitive-behaviors"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-stage-manipulation-with-demonstration","slug":"multi-stage-manipulation-with-demonstration","title":"Multi-Stage Manipulation with Demonstration-Augmented Reward, Policy, and World Model Learning","date":"2025-03-03","arxiv_id":"2503.01837","repositories_listed":1,"syntology":null},{"url":"/paper/discrete-codebook-world-models-for-continuous","slug":"discrete-codebook-world-models-for-continuous","title":"Discrete Codebook World Models for Continuous Control","date":"2025-03-01","arxiv_id":"2503.00653","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discrete-codebook-world-models-for-continuous#ran","syntology_url":"https://syntology.ai/paper/2503.00653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00653"}},"official":{"repos":["aidanscannell/dcmpc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-combinatorial-1","slug":"reinforcement-learning-with-combinatorial-1","title":"Reinforcement learning with combinatorial actions for coupled restless bandits","date":"2025-03-01","arxiv_id":"2503.01919","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-combinatorial-1#ran","syntology_url":"https://syntology.ai/paper/2503.01919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01919"}},"official":{"repos":["lily-x/combinatorial-rmab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-a-good-diffusion-planner-for","slug":"what-makes-a-good-diffusion-planner-for","title":"What Makes a Good Diffusion Planner for Decision Making?","date":"2025-03-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deepretrieval-powerful-query-generation-for","slug":"deepretrieval-powerful-query-generation-for","title":"DeepRetrieval: Hacking Real Search Engines and Retrievers with Large Language Models via Reinforcement Learning","date":"2025-02-28","arxiv_id":"2503.00223","repositories_listed":1,"syntology":{"n":25,"n_ran":24,"n_constructed":0,"n_ran_checked":23,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":23,"n_pointer_only":0,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 23 with no instrument failure: 0 honoured, 0 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepretrieval-powerful-query-generation-for#ran","syntology_url":"https://syntology.ai/paper/2503.00223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00223"}},"official":{"repos":["pat-jj/deepretrieval"],"state":"official (archive's flag): 24 ran","n_ran":24,"n_constructed":0,"n_ran_no_instrument_failure":23,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autobs-autonomous-base-station-deployment","slug":"autobs-autonomous-base-station-deployment","title":"AutoBS: Autonomous Base Station Deployment with Reinforcement Learning and Digital Network Twins","date":"2025-02-27","arxiv_id":"2502.19647","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-importance-of-reward-design-in","slug":"on-the-importance-of-reward-design-in","title":"On the Importance of Reward Design in Reinforcement Learning-based Dynamic Algorithm Configuration: A Case Study on OneMax with (1+($λ$,$λ$))-GA","date":"2025-02-27","arxiv_id":"2502.20265","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-reinforcement-learning-algorithms","slug":"distilling-reinforcement-learning-algorithms","title":"Distilling Reinforcement Learning Algorithms for In-Context Model-Based Planning","date":"2025-02-26","arxiv_id":"2502.19009","repositories_listed":1,"syntology":null},{"url":"/paper/vem-environment-free-exploration-for-training","slug":"vem-environment-free-exploration-for-training","title":"VEM: Environment-Free Exploration for Training GUI Agent with Value Environment Model","date":"2025-02-26","arxiv_id":"2502.18906","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vem-environment-free-exploration-for-training#ran","syntology_url":"https://syntology.ai/paper/2502.18906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18906"}},"official":null}},{"url":"/paper/wofostgym-a-crop-simulator-for-learning","slug":"wofostgym-a-crop-simulator-for-learning","title":"WOFOSTGym: A Crop Simulator for Learning Annual and Perennial Crop Management Strategies","date":"2025-02-26","arxiv_id":"2502.19308","repositories_listed":1,"syntology":null},{"url":"/paper/safe-multi-agent-navigation-guided-by-goal","slug":"safe-multi-agent-navigation-guided-by-goal","title":"Safe Multi-Agent Navigation guided by Goal-Conditioned Safe Reinforcement Learning","date":"2025-02-25","arxiv_id":"2502.17813","repositories_listed":1,"syntology":null},{"url":"/paper/big-math-a-large-scale-high-quality-math","slug":"big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","arxiv_id":"2502.17387","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/big-math-a-large-scale-high-quality-math#ran","syntology_url":"https://syntology.ai/paper/2502.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17387"}},"official":{"repos":["synthlabsai/big-math"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tdmpbc-self-imitative-reinforcement-learning","slug":"tdmpbc-self-imitative-reinforcement-learning","title":"TDMPBC: Self-Imitative Reinforcement Learning for Humanoid Robot Control","date":"2025-02-24","arxiv_id":"2502.17322","repositories_listed":1,"syntology":null},{"url":"/paper/statistical-inference-in-reinforcement","slug":"statistical-inference-in-reinforcement","title":"Statistical Inference in Reinforcement Learning: A Selective Survey","date":"2025-02-22","arxiv_id":"2502.16195","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-design-of-safe-continual-rl-methods","slug":"on-the-design-of-safe-continual-rl-methods","title":"On the Design of Safe Continual RL Methods for Control of Nonlinear Systems","date":"2025-02-21","arxiv_id":"2502.15922","repositories_listed":1,"syntology":null},{"url":"/paper/generating-p-functional-molecules-using-stgg","slug":"generating-p-functional-molecules-using-stgg","title":"Generating $π$-Functional Molecules Using STGG+ with Active Learning","date":"2025-02-20","arxiv_id":"2502.14842","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-reinforcement-learning-action","slug":"integrating-reinforcement-learning-action","title":"Integrating Reinforcement Learning, Action Model Learning, and Numeric Planning for Tackling Complex Tasks","date":"2025-02-18","arxiv_id":"2502.13006","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-demand-uncertainty-in-container","slug":"navigating-demand-uncertainty-in-container","title":"Navigating Demand Uncertainty in Container Shipping: Deep Reinforcement Learning for Enabling Adaptive and Feasible Master Stowage Planning","date":"2025-02-18","arxiv_id":"2502.12756","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-dynamic-resource-1","slug":"reinforcement-learning-for-dynamic-resource-1","title":"Reinforcement Learning for Dynamic Resource Allocation in Optical Networks: Hype or Hope?","date":"2025-02-18","arxiv_id":"2502.12804","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-sample-effective-and-diverse","slug":"learning-to-sample-effective-and-diverse","title":"Learning to Sample Effective and Diverse Prompts for Text-to-Image Generation","date":"2025-02-17","arxiv_id":"2502.11477","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-paperclip-maximizer-are-rl","slug":"evaluating-the-paperclip-maximizer-are-rl","title":"Evaluating the Paperclip Maximizer: Are RL-Based Language Models More Likely to Pursue Instrumental Goals?","date":"2025-02-16","arxiv_id":"2502.12206","repositories_listed":1,"syntology":null},{"url":"/paper/memory-benchmark-robots-a-benchmark-for","slug":"memory-benchmark-robots-a-benchmark-for","title":"Memory, Benchmark & Robots: A Benchmark for Solving Complex Tasks with Reinforcement Learning","date":"2025-02-14","arxiv_id":"2502.10550","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memory-benchmark-robots-a-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2502.10550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10550"}},"official":{"repos":["CognitiveAISystems/MIKASA-Robo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/digi-q-learning-q-value-functions-for","slug":"digi-q-learning-q-value-functions-for","title":"Digi-Q: Learning Q-Value Functions for Training Device-Control Agents","date":"2025-02-13","arxiv_id":"2502.15760","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/digi-q-learning-q-value-functions-for#ran","syntology_url":"https://syntology.ai/paper/2502.15760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15760"}},"official":{"repos":["digirl-agent/digiq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/hierarchical-learning-based-graph-partition","slug":"hierarchical-learning-based-graph-partition","title":"Hierarchical Learning-based Graph Partition for Large-scale Vehicle Routing Problems","date":"2025-02-12","arxiv_id":"2502.08340","repositories_listed":1,"syntology":null},{"url":"/paper/active-advantage-aligned-online-reinforcement","slug":"active-advantage-aligned-online-reinforcement","title":"Active Advantage-Aligned Online Reinforcement Learning with Offline Data","date":"2025-02-11","arxiv_id":"2502.07937","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/active-advantage-aligned-online-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07937"}},"official":{"repos":["xuefeng-cs/a3rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-view-on-learning-robust-goal-conditioned","slug":"a-view-on-learning-robust-goal-conditioned","title":"A view on learning robust goal-conditioned value functions: Interplay between RL and MPC","date":"2025-02-10","arxiv_id":"2502.06996","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-limit-of-outcome-reward-for","slug":"exploring-the-limit-of-outcome-reward-for","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","date":"2025-02-10","arxiv_id":"2502.06781","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-limit-of-outcome-reward-for#ran","syntology_url":"https://syntology.ai/paper/2502.06781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06781"}},"official":{"repos":["internlm/oreal"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-conformal-abstention-policies-for","slug":"learning-conformal-abstention-policies-for","title":"Learning Conformal Abstention Policies for Adaptive Risk Management in Large Language and Vision-Language Models","date":"2025-02-08","arxiv_id":"2502.06884","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-conformal-abstention-policies-for#ran","syntology_url":"https://syntology.ai/paper/2502.06884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06884"}},"official":{"repos":["sinatayebati/vlm-uncertainty"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/duoguard-a-two-player-rl-driven-framework-for","slug":"duoguard-a-two-player-rl-driven-framework-for","title":"DuoGuard: A Two-Player RL-Driven Framework for Multilingual LLM Guardrails","date":"2025-02-07","arxiv_id":"2502.05163","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/duoguard-a-two-player-rl-driven-framework-for#ran","syntology_url":"https://syntology.ai/paper/2502.05163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05163"}},"official":{"repos":["yihedeng9/duoguard"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/training-language-models-to-reason","slug":"training-language-models-to-reason","title":"Training Language Models to Reason Efficiently","date":"2025-02-06","arxiv_id":"2502.04463","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/training-language-models-to-reason#ran","syntology_url":"https://syntology.ai/paper/2502.04463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04463"}},"official":{"repos":["Zanette-Labs/efficient-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/ctr-driven-advertising-image-generation-with","slug":"ctr-driven-advertising-image-generation-with","title":"CTR-Driven Advertising Image Generation with Multimodal Large Language Models","date":"2025-02-05","arxiv_id":"2502.06823","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-long-chain-of-thought-reasoning","slug":"demystifying-long-chain-of-thought-reasoning","title":"Demystifying Long Chain-of-Thought Reasoning in LLMs","date":"2025-02-05","arxiv_id":"2502.03373","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/demystifying-long-chain-of-thought-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.03373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03373"}},"official":{"repos":["eddycmu/demystify-long-cot"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/underwater-soft-fin-flapping-motion-with-deep","slug":"underwater-soft-fin-flapping-motion-with-deep","title":"Underwater Soft Fin Flapping Motion with Deep Neural Network Based Surrogate Model","date":"2025-02-05","arxiv_id":"2502.03135","repositories_listed":1,"syntology":null},{"url":"/paper/analytical-lyapunov-function-discovery-an-rl","slug":"analytical-lyapunov-function-discovery-an-rl","title":"Analytical Lyapunov Function Discovery: An RL-based Generative Approach","date":"2025-02-04","arxiv_id":"2502.02014","repositories_listed":1,"syntology":null},{"url":"/paper/circular-microalgae-based-carbon-control-for","slug":"circular-microalgae-based-carbon-control-for","title":"Circular Microalgae-Based Carbon Control for Net Zero","date":"2025-02-04","arxiv_id":"2502.02382","repositories_listed":1,"syntology":null},{"url":"/paper/reusing-embeddings-reproducible-reward-model","slug":"reusing-embeddings-reproducible-reward-model","title":"Reusing Embeddings: Reproducible Reward Model Research in Large Language Model Alignment without GPUs","date":"2025-02-04","arxiv_id":"2502.04357","repositories_listed":1,"syntology":null},{"url":"/paper/gnn-dt-graph-neural-network-enhanced-decision","slug":"gnn-dt-graph-neural-network-enhanced-decision","title":"GNN-DT: Graph Neural Network Enhanced Decision Transformer for Efficient Optimization in Dynamic Environments","date":"2025-02-03","arxiv_id":"2502.01778","repositories_listed":1,"syntology":null},{"url":"/paper/recursive-generalized-type-2-fuzzy-radial","slug":"recursive-generalized-type-2-fuzzy-radial","title":"Recursive generalized type-2 fuzzy radial basis function neural networks for joint position estimation and adaptive EMG-based impedance control of lower limb exoskeletons","date":"2025-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sharpie-a-modular-framework-for-reinforcement","slug":"sharpie-a-modular-framework-for-reinforcement","title":"SHARPIE: A Modular Framework for Reinforcement Learning and Human-AI Interaction Experiments","date":"2025-01-31","arxiv_id":"2501.19245","repositories_listed":1,"syntology":null},{"url":"/paper/test-time-training-scaling-for-chemical","slug":"test-time-training-scaling-for-chemical","title":"Test-Time Training Scaling Laws for Chemical Exploration in Drug Design","date":"2025-01-31","arxiv_id":"2501.19153","repositories_listed":1,"syntology":null},{"url":"/paper/neural-operator-based-reinforcement-learning","slug":"neural-operator-based-reinforcement-learning","title":"Neural Operator based Reinforcement Learning for Control of first-order PDEs with Spatially-Varying State Delay","date":"2025-01-30","arxiv_id":"2501.18201","repositories_listed":1,"syntology":null},{"url":"/paper/langevin-soft-actor-critic-efficient","slug":"langevin-soft-actor-critic-efficient","title":"Langevin Soft Actor-Critic: Efficient Exploration through Uncertainty-Driven Critic Learning","date":"2025-01-29","arxiv_id":"2501.17827","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/langevin-soft-actor-critic-efficient#ran","syntology_url":"https://syntology.ai/paper/2501.17827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17827"}},"official":{"repos":["hmishfaq/lsac"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/rlpp-a-residual-method-for-zero-shot-real","slug":"rlpp-a-residual-method-for-zero-shot-real","title":"RLPP: A Residual Method for Zero-Shot Real-World Autonomous Racing on Scaled Platforms","date":"2025-01-28","arxiv_id":"2501.17311","repositories_listed":1,"syntology":null},{"url":"/paper/xjailbreak-representation-space-guided","slug":"xjailbreak-representation-space-guided","title":"xJailbreak: Representation Space Guided Reinforcement Learning for Interpretable LLM Jailbreaking","date":"2025-01-28","arxiv_id":"2501.16727","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-quantum-reinforcement-learning","slug":"benchmarking-quantum-reinforcement-learning","title":"Benchmarking Quantum Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.15893","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2501.15893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15893"}},"official":{"repos":["nicomeyer96/qrl-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/expert-free-online-transfer-learning-in-multi-1","slug":"expert-free-online-transfer-learning-in-multi-1","title":"Expert-Free Online Transfer Learning in Multi-Agent Reinforcement Learning","date":"2025-01-26","arxiv_id":"2501.15495","repositories_listed":1,"syntology":null},{"url":"/paper/improving-retrieval-augmented-generation","slug":"improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15228","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2501.15228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15228"}},"official":{"repos":["chenyiqun/mmoa-rag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-attentive-graph-agent-for-topology","slug":"an-attentive-graph-agent-for-topology","title":"An Attentive Graph Agent for Topology-Adaptive Cyber Defence","date":"2025-01-24","arxiv_id":"2501.14700","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-data-exploitation-in-deep","slug":"adaptive-data-exploitation-in-deep","title":"Adaptive Data Exploitation in Deep Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12620","repositories_listed":1,"syntology":null},{"url":"/paper/to-measure-or-not-a-cost-sensitive-selective","slug":"to-measure-or-not-a-cost-sensitive-selective","title":"To Measure or Not: A Cost-Sensitive, Selective Measuring Environment for Agricultural Management Decisions with Reinforcement Learning","date":"2025-01-22","arxiv_id":"2501.12823","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-language-model-reasoning-through","slug":"advancing-language-model-reasoning-through","title":"Advancing Language Model Reasoning through Reinforcement Learning and Inference Scaling","date":"2025-01-20","arxiv_id":"2501.11651","repositories_listed":1,"syntology":null},{"url":"/paper/improving-thermal-state-preparation-of","slug":"improving-thermal-state-preparation-of","title":"Improving thermal state preparation of Sachdev-Ye-Kitaev model with reinforcement learning on quantum hardware","date":"2025-01-20","arxiv_id":"2501.11454","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-language-models-a-blueprint","slug":"reasoning-language-models-a-blueprint","title":"Reasoning Language Models: A Blueprint","date":"2025-01-20","arxiv_id":"2501.11223","repositories_listed":1,"syntology":null},{"url":"/paper/green-code-optimizing-energy-efficiency-in","slug":"green-code-optimizing-energy-efficiency-in","title":"GREEN-CODE: Learning to Optimize Energy Efficiency in LLM-based Code Generation","date":"2025-01-19","arxiv_id":"2501.11006","repositories_listed":1,"syntology":null},{"url":"/paper/pixelbrax-learning-continuous-control-from","slug":"pixelbrax-learning-continuous-control-from","title":"PixelBrax: Learning Continuous Control from Pixels End-to-End on the GPU","date":"2025-01-16","arxiv_id":"2502.00021","repositories_listed":1,"syntology":null},{"url":"/paper/cheq-ing-the-box-safe-variable-impedance","slug":"cheq-ing-the-box-safe-variable-impedance","title":"CHEQ-ing the Box: Safe Variable Impedance Learning for Robotic Polishing","date":"2025-01-14","arxiv_id":"2501.07985","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-online-reinforcement-learning-with","slug":"enhancing-online-reinforcement-learning-with","title":"Enhancing Online Reinforcement Learning with Meta-Learned Objective from Offline Data","date":"2025-01-13","arxiv_id":"2501.07346","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-deep-reinforcement","slug":"an-empirical-study-of-deep-reinforcement","title":"An Empirical Study of Deep Reinforcement Learning in Continuing Tasks","date":"2025-01-12","arxiv_id":"2501.06937","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-framework-for-reinsurance","slug":"a-hybrid-framework-for-reinsurance","title":"A Hybrid Framework for Reinsurance Optimization: Integrating Generative Models and Reinforcement Learning","date":"2025-01-11","arxiv_id":"2501.06404","repositories_listed":1,"syntology":null},{"url":"/paper/from-discrete-time-policies-to-continuous","slug":"from-discrete-time-policies-to-continuous","title":"From discrete-time policies to continuous-time diffusion samplers: Asymptotic equivalences and faster training","date":"2025-01-10","arxiv_id":"2501.06148","repositories_listed":1,"syntology":null},{"url":"/paper/smart-imitator-learning-from-imperfect","slug":"smart-imitator-learning-from-imperfect","title":"Smart Imitator: Learning from Imperfect Clinical Decisions","date":"2025-01-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multilinear-tensor-low-rank-approximation-for","slug":"multilinear-tensor-low-rank-approximation-for","title":"Multilinear Tensor Low-Rank Approximation for Policy-Gradient Methods in Reinforcement Learning","date":"2025-01-08","arxiv_id":"2501.04879","repositories_listed":1,"syntology":null},{"url":"/paper/co-activation-graph-analysis-of-safety","slug":"co-activation-graph-analysis-of-safety","title":"Co-Activation Graph Analysis of Safety-Verified and Explainable Deep Reinforcement Learning Policies","date":"2025-01-06","arxiv_id":"2501.03142","repositories_listed":1,"syntology":null},{"url":"/paper/digital-twin-aided-channel-estimation-zone","slug":"digital-twin-aided-channel-estimation-zone","title":"Digital Twin Aided Channel Estimation: Zone-Specific Subspace Prediction and Calibration","date":"2025-01-06","arxiv_id":"2501.02758","repositories_listed":1,"syntology":null},{"url":"/paper/sim-to-real-transfer-for-mobile-robots-with","slug":"sim-to-real-transfer-for-mobile-robots-with","title":"Sim-to-Real Transfer for Mobile Robots with Reinforcement Learning: from NVIDIA Isaac Sim to Gazebo and Real ROS 2 Robots","date":"2025-01-06","arxiv_id":"2501.02902","repositories_listed":1,"syntology":null},{"url":"/paper/the-meta-representation-hypothesis","slug":"the-meta-representation-hypothesis","title":"Representation Convergence: Mutual Distillation is Secretly a Form of Regularization","date":"2025-01-05","arxiv_id":"2501.02481","repositories_listed":1,"syntology":null},{"url":"/paper/noise-resilient-symbolic-regression-with","slug":"noise-resilient-symbolic-regression-with","title":"Noise-Resilient Symbolic Regression with Dynamic Gating Reinforcement Learning","date":"2025-01-02","arxiv_id":"2501.01085","repositories_listed":1,"syntology":null},{"url":"/paper/hybridising-reinforcement-learning-and","slug":"hybridising-reinforcement-learning-and","title":"Hybridising Reinforcement Learning and Heuristics for Hierarchical Directed Arc Routing Problems","date":"2025-01-01","arxiv_id":"2501.00852","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-hybrid-policy-in-reinforcement","slug":"exploiting-hybrid-policy-in-reinforcement","title":"Exploiting Hybrid Policy in Reinforcement Learning for Interpretable Temporal Logic Manipulation","date":"2024-12-29","arxiv_id":"2412.20338","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-scalable-deep-reinforcement","slug":"efficient-and-scalable-deep-reinforcement","title":"Efficient and Scalable Deep Reinforcement Learning for Mean Field Control Games","date":"2024-12-28","arxiv_id":"2501.00052","repositories_listed":1,"syntology":null},{"url":"/paper/xsrl-safety-aware-explainable-reinforcement","slug":"xsrl-safety-aware-explainable-reinforcement","title":"xSRL: Safety-Aware Explainable Reinforcement Learning -- Safety as a Product of Explainability","date":"2024-12-26","arxiv_id":"2412.19311","repositories_listed":1,"syntology":null},{"url":"/paper/huatuogpt-o1-towards-medical-complex","slug":"huatuogpt-o1-towards-medical-complex","title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","date":"2024-12-25","arxiv_id":"2412.18925","repositories_listed":1,"syntology":null}],"record_sha256":"3b8bde59d118934582fd5642771fe4518922ec1b5d5da3df8197d670e58e91af","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}