{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/12","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":152,"rows_per_page":100,"rows":[1101,1200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/11","next":"/task/reinforcement-learning-1/papers/13","papers":[{"url":"/paper/extrans-multilingual-deep-reasoning","slug":"extrans-multilingual-deep-reasoning","title":"ExTrans: Multilingual Deep Reasoning Translation via Exemplar-Enhanced Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.12996","repositories_listed":1,"syntology":null},{"url":"/paper/g1-bootstrapping-perception-and-reasoning","slug":"g1-bootstrapping-perception-and-reasoning","title":"G1: Bootstrapping Perception and Reasoning Abilities of Vision-Language Model via Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.13426","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-anytime-reasoning-via-budget","slug":"optimizing-anytime-reasoning-via-budget","title":"Optimizing Anytime Reasoning via Budget Relative Policy Optimization","date":"2025-05-19","arxiv_id":"2505.13438","repositories_listed":1,"syntology":null},{"url":"/paper/cpgd-toward-stable-rule-based-reinforcement","slug":"cpgd-toward-stable-rule-based-reinforcement","title":"CPGD: Toward Stable Rule-based Reinforcement Learning for Language Models","date":"2025-05-18","arxiv_id":"2505.12504","repositories_listed":1,"syntology":null},{"url":"/paper/graph-reward-sql-execution-free-reinforcement","slug":"graph-reward-sql-execution-free-reinforcement","title":"Graph-Reward-SQL: Execution-Free Reinforcement Learning for Text-to-SQL via Graph Matching and Stepwise Reward","date":"2025-05-18","arxiv_id":"2505.12380","repositories_listed":1,"syntology":null},{"url":"/paper/observe-r1-unlocking-reasoning-abilities-of","slug":"observe-r1-unlocking-reasoning-abilities-of","title":"Observe-R1: Unlocking Reasoning Abilities of MLLMs with Dynamic Progressive Reinforcement Learning","date":"2025-05-18","arxiv_id":"2505.12432","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-data-rl-task-definition-is-all-you","slug":"synthetic-data-rl-task-definition-is-all-you","title":"Synthetic Data RL: Task Definition Is All You Need","date":"2025-05-18","arxiv_id":"2505.17063","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/synthetic-data-rl-task-definition-is-all-you#ran","syntology_url":"https://syntology.ai/paper/2505.17063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17063"}},"official":{"repos":["gydpku/data_synthesis_rl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/videorft-incentivizing-video-reasoning","slug":"videorft-incentivizing-video-reasoning","title":"VideoRFT: Incentivizing Video Reasoning Capability in MLLMs via Reinforced Fine-Tuning","date":"2025-05-18","arxiv_id":"2505.12434","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videorft-incentivizing-video-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.12434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12434"}},"official":{"repos":["qiwang98/videorft"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/retrospex-language-agent-meets-offline","slug":"retrospex-language-agent-meets-offline","title":"Retrospex: Language Agent Meets Offline Reinforcement Learning Critic","date":"2025-05-17","arxiv_id":"2505.11807","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10802","slug":"2505-10802","title":"Attention-Based Reward Shaping for Sparse and Delayed Rewards","date":"2025-05-16","arxiv_id":"2505.10802","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10861","slug":"2505-10861","title":"Improving the Data-efficiency of Reinforcement Learning by Warm-starting with LLM","date":"2025-05-16","arxiv_id":"2505.10861","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10861#ran","syntology_url":"https://syntology.ai/paper/2505.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10861"}},"official":{"repos":["duongnhatthang/llamagym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10978","slug":"2505-10978","title":"Group-in-Group Policy Optimization for LLM Agent Training","date":"2025-05-16","arxiv_id":"2505.10978","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11032","slug":"2505-11032","title":"DexGarmentLab: Dexterous Garment Manipulation Environment with Generalizable Policy","date":"2025-05-16","arxiv_id":"2505.11032","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11221","slug":"2505-11221","title":"Sample Efficient Reinforcement Learning via Large Vision Language Model Distillation","date":"2025-05-16","arxiv_id":"2505.11221","repositories_listed":1,"syntology":null},{"url":"/paper/an-agentic-system-with-reinforcement-learned","slug":"an-agentic-system-with-reinforcement-learned","title":"An agentic system with reinforcement-learned subsystem improvements for parsing form-like documents","date":"2025-05-16","arxiv_id":"2505.13504","repositories_listed":1,"syntology":null},{"url":"/paper/time-r1-towards-comprehensive-temporal","slug":"time-r1-towards-comprehensive-temporal","title":"Time-R1: Towards Comprehensive Temporal Reasoning in LLMs","date":"2025-05-16","arxiv_id":"2505.13508","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-aha-toward-systematic-meta-abilities","slug":"beyond-aha-toward-systematic-meta-abilities","title":"Beyond 'Aha!': Toward Systematic Meta-Abilities Alignment in Large Reasoning Models","date":"2025-05-15","arxiv_id":"2505.10554","repositories_listed":1,"syntology":null},{"url":"/paper/imaginebench-evaluating-reinforcement","slug":"imaginebench-evaluating-reinforcement","title":"ImagineBench: Evaluating Reinforcement Learning with Large Language Model Rollouts","date":"2025-05-15","arxiv_id":"2505.10010","repositories_listed":1,"syntology":null},{"url":"/paper/openthinkimg-learning-to-think-with-images","slug":"openthinkimg-learning-to-think-with-images","title":"OpenThinkIMG: Learning to Think with Images via Visual Tool Reinforcement Learning","date":"2025-05-13","arxiv_id":"2505.08617","repositories_listed":1,"syntology":null},{"url":"/paper/dancegrpo-unleashing-grpo-on-visual","slug":"dancegrpo-unleashing-grpo-on-visual","title":"DanceGRPO: Unleashing GRPO on Visual Generation","date":"2025-05-12","arxiv_id":"2505.07818","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dancegrpo-unleashing-grpo-on-visual#ran","syntology_url":"https://syntology.ai/paper/2505.07818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07818"}},"official":null}},{"url":"/paper/dynamicrag-leveraging-outputs-of-large","slug":"dynamicrag-leveraging-outputs-of-large","title":"DynamicRAG: Leveraging Outputs of Large Language Model as Feedback for Dynamic Reranking in Retrieval-Augmented Generation","date":"2025-05-12","arxiv_id":"2505.07233","repositories_listed":1,"syntology":null},{"url":"/paper/kalman-filter-enhanced-grpo-for-reinforcement","slug":"kalman-filter-enhanced-grpo-for-reinforcement","title":"Kalman Filter Enhanced GRPO for Reinforcement Learning-Based Language Model Reasoning","date":"2025-05-12","arxiv_id":"2505.07527","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-general-intelligence-with-generated","slug":"measuring-general-intelligence-with-generated","title":"Measuring General Intelligence with Generated Games","date":"2025-05-12","arxiv_id":"2505.07215","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-general-intelligence-with-generated#ran","syntology_url":"https://syntology.ai/paper/2505.07215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07215"}},"official":{"repos":["vivek3141/gg-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforced-internal-external-knowledge","slug":"reinforced-internal-external-knowledge","title":"Reinforced Internal-External Knowledge Synergistic Reasoning for Efficient Adaptive Search Agent","date":"2025-05-12","arxiv_id":"2505.07596","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforced-internal-external-knowledge#ran","syntology_url":"https://syntology.ai/paper/2505.07596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07596"}},"official":null}},{"url":"/paper/lineflow-a-framework-to-learn-active-control","slug":"lineflow-a-framework-to-learn-active-control","title":"LineFlow: A Framework to Learn Active Control of Production Lines","date":"2025-05-10","arxiv_id":"2505.06744","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lineflow-a-framework-to-learn-active-control#ran","syntology_url":"https://syntology.ai/paper/2505.06744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.06744"}},"official":{"repos":["hs-kempten/lineflow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-grpo-training-flow-matching-models-via","slug":"flow-grpo-training-flow-matching-models-via","title":"Flow-GRPO: Training Flow Matching Models via Online RL","date":"2025-05-08","arxiv_id":"2505.05470","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flow-grpo-training-flow-matching-models-via#ran","syntology_url":"https://syntology.ai/paper/2505.05470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05470"}},"official":{"repos":["yifan123/flow_grpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-autonomous-cyber","slug":"large-language-models-are-autonomous-cyber","title":"Large Language Models are Autonomous Cyber Defenders","date":"2025-05-07","arxiv_id":"2505.04843","repositories_listed":1,"syntology":null},{"url":"/paper/zerosearch-incentivize-the-search-capability","slug":"zerosearch-incentivize-the-search-capability","title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","date":"2025-05-07","arxiv_id":"2505.04588","repositories_listed":1,"syntology":{"n":15,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/zerosearch-incentivize-the-search-capability#ran","syntology_url":"https://syntology.ai/paper/2505.04588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04588"}},"official":{"repos":["alibaba-nlp/zerosearch"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/emorl-ensemble-multi-objective-reinforcement","slug":"emorl-ensemble-multi-objective-reinforcement","title":"EMORL: Ensemble Multi-Objective Reinforcement Learning for Efficient and Flexible LLM Fine-Tuning","date":"2025-05-05","arxiv_id":"2505.02579","repositories_listed":1,"syntology":null},{"url":"/paper/r1-reward-training-multimodal-reward-model","slug":"r1-reward-training-multimodal-reward-model","title":"R1-Reward: Training Multimodal Reward Model Through Stable Reinforcement Learning","date":"2025-05-05","arxiv_id":"2505.02835","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/r1-reward-training-multimodal-reward-model#ran","syntology_url":"https://syntology.ai/paper/2505.02835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02835"}},"official":{"repos":["yfzhang114/r1_reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rm-r1-reward-modeling-as-reasoning","slug":"rm-r1-reward-modeling-as-reasoning","title":"RM-R1: Reward Modeling as Reasoning","date":"2025-05-05","arxiv_id":"2505.02387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/rm-r1-reward-modeling-as-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.02387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02387"}},"official":{"repos":["rm-r1-uiuc/rm-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-generalised-and-adaptable-reinforcement","slug":"a-generalised-and-adaptable-reinforcement","title":"A Generalised and Adaptable Reinforcement Learning Stopping Method","date":"2025-05-03","arxiv_id":"2505.01907","repositories_listed":1,"syntology":null},{"url":"/paper/directly-forecasting-belief-for-reinforcement","slug":"directly-forecasting-belief-for-reinforcement","title":"Directly Forecasting Belief for Reinforcement Learning with Delays","date":"2025-05-01","arxiv_id":"2505.00546","repositories_listed":1,"syntology":null},{"url":"/paper/smallplan-leverage-small-language-models-for","slug":"smallplan-leverage-small-language-models-for","title":"SmallPlan: Leverage Small Language Models for Sequential Path Planning with Simulation-Powered, LLM-Guided Distillation","date":"2025-05-01","arxiv_id":"2505.00831","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-new-item-fairness-in-dynamic","slug":"enhancing-new-item-fairness-in-dynamic","title":"Enhancing New-item Fairness in Dynamic Recommender Systems","date":"2025-04-30","arxiv_id":"2504.21362","repositories_listed":1,"syntology":null},{"url":"/paper/rulebook-bringing-co-routines-to","slug":"rulebook-bringing-co-routines-to","title":"Rulebook: bringing co-routines to reinforcement learning environments","date":"2025-04-28","arxiv_id":"2504.19625","repositories_listed":1,"syntology":null},{"url":"/paper/bqsched-a-non-intrusive-scheduler-for-batch","slug":"bqsched-a-non-intrusive-scheduler-for-batch","title":"BQSched: A Non-intrusive Scheduler for Batch Concurrent Queries via Reinforcement Learning","date":"2025-04-27","arxiv_id":"2504.19142","repositories_listed":1,"syntology":null},{"url":"/paper/neurophysiologically-realistic-environment","slug":"neurophysiologically-realistic-environment","title":"Neurophysiologically Realistic Environment for Comparing Adaptive Deep Brain Stimulation Algorithms in Parkinson Disease","date":"2025-04-26","arxiv_id":"2505.09624","repositories_listed":1,"syntology":null},{"url":"/paper/carl-learning-scalable-planning-policies-with","slug":"carl-learning-scalable-planning-policies-with","title":"CaRL: Learning Scalable Planning Policies with Simple Rewards","date":"2025-04-24","arxiv_id":"2504.17838","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/carl-learning-scalable-planning-policies-with#ran","syntology_url":"https://syntology.ai/paper/2504.17838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.17838"}},"official":{"repos":["autonomousvision/CaRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tina-tiny-reasoning-models-via-lora","slug":"tina-tiny-reasoning-models-via-lora","title":"Tina: Tiny Reasoning Models via LoRA","date":"2025-04-22","arxiv_id":"2504.15777","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tina-tiny-reasoning-models-via-lora#ran","syntology_url":"https://syntology.ai/paper/2504.15777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15777"}},"official":{"repos":["shangshang-wang/tina"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/flowreasoner-reinforcing-query-level-meta","slug":"flowreasoner-reinforcing-query-level-meta","title":"FlowReasoner: Reinforcing Query-Level Meta-Agents","date":"2025-04-21","arxiv_id":"2504.15257","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reason-under-off-policy-guidance","slug":"learning-to-reason-under-off-policy-guidance","title":"Learning to Reason under Off-Policy Guidance","date":"2025-04-21","arxiv_id":"2504.14945","repositories_listed":1,"syntology":null},{"url":"/paper/stop-summation-min-form-credit-assignment-is","slug":"stop-summation-min-form-credit-assignment-is","title":"Stop Summation: Min-Form Credit Assignment Is All Process Reward Model Needs for Reasoning","date":"2025-04-21","arxiv_id":"2504.15275","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stop-summation-min-form-credit-assignment-is#ran","syntology_url":"https://syntology.ai/paper/2504.15275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15275"}},"official":{"repos":["cjreinforce/pure"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-auto-bidding-with-value-guided","slug":"generative-auto-bidding-with-value-guided","title":"Generative Auto-Bidding with Value-Guided Explorations","date":"2025-04-20","arxiv_id":"2504.14587","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generative-auto-bidding-with-value-guided#ran","syntology_url":"https://syntology.ai/paper/2504.14587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14587"}},"official":{"repos":["applied-machine-learning-lab/gave"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/compile-scene-graphs-with-reinforcement","slug":"compile-scene-graphs-with-reinforcement","title":"Compile Scene Graphs with Reinforcement Learning","date":"2025-04-18","arxiv_id":"2504.13617","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compile-scene-graphs-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.13617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13617"}},"official":{"repos":["gpt4vision/r1-sgg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prejudge-before-think-enhancing-large","slug":"prejudge-before-think-enhancing-large","title":"Prejudge-Before-Think: Enhancing Large Language Models at Test-Time by Process Prejudge Reasoning","date":"2025-04-18","arxiv_id":"2504.13500","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-r-collaborative-framework-for","slug":"embodied-r-collaborative-framework-for","title":"Embodied-R: Collaborative Framework for Activating Embodied Spatial Reasoning in Foundation Models via Reinforcement Learning","date":"2025-04-17","arxiv_id":"2504.12680","repositories_listed":1,"syntology":null},{"url":"/paper/noisyrollout-reinforcing-visual-reasoning","slug":"noisyrollout-reinforcing-visual-reasoning","title":"NoisyRollout: Reinforcing Visual Reasoning with Data Augmentation","date":"2025-04-17","arxiv_id":"2504.13055","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/noisyrollout-reinforcing-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2504.13055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13055"}},"official":null}},{"url":"/paper/skyreels-v2-infinite-length-film-generative","slug":"skyreels-v2-infinite-length-film-generative","title":"SkyReels-V2: Infinite-length Film Generative Model","date":"2025-04-17","arxiv_id":"2504.13074","repositories_listed":1,"syntology":null},{"url":"/paper/control-of-rayleigh-benard-convection","slug":"control-of-rayleigh-benard-convection","title":"Control of Rayleigh-Bénard Convection: Effectiveness of Reinforcement Learning in the Turbulent Regime","date":"2025-04-16","arxiv_id":"2504.12000","repositories_listed":1,"syntology":null},{"url":"/paper/toolrl-reward-is-all-tool-learning-needs","slug":"toolrl-reward-is-all-tool-learning-needs","title":"ToolRL: Reward is All Tool Learning Needs","date":"2025-04-16","arxiv_id":"2504.13958","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toolrl-reward-is-all-tool-learning-needs#ran","syntology_url":"https://syntology.ai/paper/2504.13958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13958"}},"official":{"repos":["qiancheng0/toolrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-minimalist-approach-to-llm-reasoning-from","slug":"a-minimalist-approach-to-llm-reasoning-from","title":"A Minimalist Approach to LLM Reasoning: from Rejection Sampling to Reinforce","date":"2025-04-15","arxiv_id":"2504.11343","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-minimalist-approach-to-llm-reasoning-from#ran","syntology_url":"https://syntology.ai/paper/2504.11343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11343"}},"official":{"repos":["rlhflow/minimal-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-driven-approach-towards-more-efficient","slug":"data-driven-approach-towards-more-efficient","title":"Data driven approach towards more efficient Newton-Raphson power flow calculation for distribution grids","date":"2025-04-15","arxiv_id":"2504.11650","repositories_listed":1,"syntology":null},{"url":"/paper/deepmath-103k-a-large-scale-challenging","slug":"deepmath-103k-a-large-scale-challenging","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","date":"2025-04-15","arxiv_id":"2504.11456","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepmath-103k-a-large-scale-challenging#ran","syntology_url":"https://syntology.ai/paper/2504.11456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11456"}},"official":{"repos":["zwhe99/deepmath"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/retool-reinforcement-learning-for-strategic","slug":"retool-reinforcement-learning-for-strategic","title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","date":"2025-04-15","arxiv_id":"2504.11536","repositories_listed":1,"syntology":null},{"url":"/paper/mt-r1-zero-advancing-llm-based-machine","slug":"mt-r1-zero-advancing-llm-based-machine","title":"MT-R1-Zero: Advancing LLM-based Machine Translation via R1-Zero-like Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10160","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mt-r1-zero-advancing-llm-based-machine#ran","syntology_url":"https://syntology.ai/paper/2504.10160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10160"}},"official":{"repos":["fzp0424/mt-r1-zero"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dump-automated-distribution-level-curriculum","slug":"dump-automated-distribution-level-curriculum","title":"DUMP: Automated Distribution-Level Curriculum Learning for RL-based LLM Post-training","date":"2025-04-13","arxiv_id":"2504.09710","repositories_listed":1,"syntology":null},{"url":"/paper/development-of-a-ppo-reinforcement-learned","slug":"development-of-a-ppo-reinforcement-learned","title":"Development of a PPO-Reinforcement Learned Walking Tripedal Soft-Legged Robot using SOFA","date":"2025-04-12","arxiv_id":"2504.09242","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-power-grid-topologies-with","slug":"optimizing-power-grid-topologies-with","title":"Optimizing Power Grid Topologies with Reinforcement Learning: A Survey of Methods and Challenges","date":"2025-04-11","arxiv_id":"2504.08210","repositories_listed":1,"syntology":null},{"url":"/paper/echo-chamber-rl-post-training-amplifies","slug":"echo-chamber-rl-post-training-amplifies","title":"Echo Chamber: RL Post-training Amplifies Behaviors Learned in Pretraining","date":"2025-04-10","arxiv_id":"2504.07912","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/echo-chamber-rl-post-training-amplifies#ran","syntology_url":"https://syntology.ai/paper/2504.07912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07912"}},"official":{"repos":["rosieyzh/openrlhf-pretrain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-equivariance-modeling-turbulence","slug":"harnessing-equivariance-modeling-turbulence","title":"Harnessing Equivariance: Modeling Turbulence with Graph Neural Networks","date":"2025-04-10","arxiv_id":"2504.07741","repositories_listed":1,"syntology":null},{"url":"/paper/kimi-vl-technical-report","slug":"kimi-vl-technical-report","title":"Kimi-VL Technical Report","date":"2025-04-10","arxiv_id":"2504.07491","repositories_listed":1,"syntology":null},{"url":"/paper/perception-r1-pioneering-perception-policy","slug":"perception-r1-pioneering-perception-policy","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.07954","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/perception-r1-pioneering-perception-policy#ran","syntology_url":"https://syntology.ai/paper/2504.07954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07954"}},"official":{"repos":["linkangheng/pr1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sft-or-rl-an-early-investigation-into","slug":"sft-or-rl-an-early-investigation-into","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","date":"2025-04-10","arxiv_id":"2504.11468","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/sft-or-rl-an-early-investigation-into#ran","syntology_url":"https://syntology.ai/paper/2504.11468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11468"}},"official":null}},{"url":"/paper/vlm-r1-a-stable-and-generalizable-r1-style","slug":"vlm-r1-a-stable-and-generalizable-r1-style","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","date":"2025-04-10","arxiv_id":"2504.07615","repositories_listed":1,"syntology":null},{"url":"/paper/neural-motion-simulator-pushing-the-limit-of","slug":"neural-motion-simulator-pushing-the-limit-of","title":"Neural Motion Simulator: Pushing the Limit of World Models in Reinforcement Learning","date":"2025-04-09","arxiv_id":"2504.07095","repositories_listed":1,"syntology":null},{"url":"/paper/right-question-is-already-half-the-answer","slug":"right-question-is-already-half-the-answer","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","date":"2025-04-08","arxiv_id":"2504.05812","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/right-question-is-already-half-the-answer#ran","syntology_url":"https://syntology.ai/paper/2504.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05812"}},"official":{"repos":["qingyangzhang/empo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/trust-region-twisted-policy-improvement","slug":"trust-region-twisted-policy-improvement","title":"Trust-Region Twisted Policy Improvement","date":"2025-04-08","arxiv_id":"2504.06048","repositories_listed":1,"syntology":null},{"url":"/paper/concise-reasoning-via-reinforcement-learning","slug":"concise-reasoning-via-reinforcement-learning","title":"Concise Reasoning via Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05185","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/concise-reasoning-via-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05185"}},"official":{"repos":["ai-wand/concise-reasoning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/joint-pedestrian-and-vehicle-traffic","slug":"joint-pedestrian-and-vehicle-traffic","title":"Joint Pedestrian and Vehicle Traffic Optimization in Urban Environments using Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05018","repositories_listed":1,"syntology":null},{"url":"/paper/deepresearcher-scaling-deep-research-via","slug":"deepresearcher-scaling-deep-research-via","title":"DeepResearcher: Scaling Deep Research via Reinforcement Learning in Real-world Environments","date":"2025-04-04","arxiv_id":"2504.03160","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepresearcher-scaling-deep-research-via#ran","syntology_url":"https://syntology.ai/paper/2504.03160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.03160"}},"official":{"repos":["gair-nlp/deepresearcher"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpg-a-simple-and-strong-reinforcement","slug":"gpg-a-simple-and-strong-reinforcement","title":"GPG: A Simple and Strong Reinforcement Learning Baseline for Model Reasoning","date":"2025-04-03","arxiv_id":"2504.02546","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpg-a-simple-and-strong-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.02546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02546"}},"official":{"repos":["amap-ml/gpg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mad-a-magnitude-and-direction-policy","slug":"mad-a-magnitude-and-direction-policy","title":"MAD: A Magnitude And Direction Policy Parametrization for Stability Constrained Reinforcement Learning","date":"2025-04-03","arxiv_id":"2504.02565","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-under-1-billion-memory-augmented","slug":"reasoning-under-1-billion-memory-augmented","title":"Reasoning Under 1 Billion: Memory-Augmented Reinforcement Learning for Large Language Models","date":"2025-04-03","arxiv_id":"2504.02273","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-rl-scaling-for-vision-language","slug":"rethinking-rl-scaling-for-vision-language","title":"Rethinking RL Scaling for Vision Language Models: A Transparent, From-Scratch Framework and Comprehensive Evaluation Scheme","date":"2025-04-03","arxiv_id":"2504.02587","repositories_listed":1,"syntology":null},{"url":"/paper/gmai-vl-r1-harnessing-reinforcement-learning","slug":"gmai-vl-r1-harnessing-reinforcement-learning","title":"GMAI-VL-R1: Harnessing Reinforcement Learning for Multimodal Medical Reasoning","date":"2025-04-02","arxiv_id":"2504.01886","repositories_listed":1,"syntology":null},{"url":"/paper/thinkprune-pruning-long-chain-of-thought-of","slug":"thinkprune-pruning-long-chain-of-thought-of","title":"ThinkPrune: Pruning Long Chain-of-Thought of LLMs via Reinforcement Learning","date":"2025-04-02","arxiv_id":"2504.01296","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/thinkprune-pruning-long-chain-of-thought-of#ran","syntology_url":"https://syntology.ai/paper/2504.01296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01296"}},"official":{"repos":["UCSB-NLP-Chang/ThinkPrune"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/tom-rl-reinforcement-learning-unlocks-theory","slug":"tom-rl-reinforcement-learning-unlocks-theory","title":"Do Theory of Mind Benchmarks Need Explicit Human-like Reasoning in Language Models?","date":"2025-04-02","arxiv_id":"2504.01698","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tom-rl-reinforcement-learning-unlocks-theory#ran","syntology_url":"https://syntology.ai/paper/2504.01698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01698"}},"official":{"repos":["bigai-ai/ToM-RL"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mpcritic-a-plug-and-play-mpc-architecture-for","slug":"mpcritic-a-plug-and-play-mpc-architecture-for","title":"MPCritic: A plug-and-play MPC architecture for reinforcement learning","date":"2025-04-01","arxiv_id":"2504.01086","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistically-safe-and-efficient-model","slug":"probabilistically-safe-and-efficient-model","title":"Probabilistically safe and efficient model-based Reinforcement Learning","date":"2025-04-01","arxiv_id":"2504.00626","repositories_listed":1,"syntology":null},{"url":"/paper/value-iteration-for-learning-concurrently","slug":"value-iteration-for-learning-concurrently","title":"Value Iteration for Learning Concurrently Executable Robotic Control Tasks","date":"2025-04-01","arxiv_id":"2504.01174","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-effect-of-reinforcement","slug":"exploring-the-effect-of-reinforcement","title":"Exploring the Effect of Reinforcement Learning on Video Understanding: Insights from SEED-Bench-R1","date":"2025-03-31","arxiv_id":"2503.24376","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-the-effect-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.24376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24376"}},"official":{"repos":["tencentarc/seed-bench-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/handling-delay-in-real-time-reinforcement","slug":"handling-delay-in-real-time-reinforcement","title":"Handling Delay in Real-Time Reinforcement Learning","date":"2025-03-30","arxiv_id":"2503.23478","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/handling-delay-in-real-time-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.23478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23478"}},"official":{"repos":["avecplezir/realtime-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/reinforcement-learning-based-token-pruning-in","slug":"reinforcement-learning-based-token-pruning-in","title":"Reinforcement Learning-based Token Pruning in Vision Transformers: A Markov Game Approach","date":"2025-03-30","arxiv_id":"2503.23459","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-large-language-model-with-latent","slug":"controlling-large-language-model-with-latent","title":"Controlling Large Language Model with Latent Actions","date":"2025-03-27","arxiv_id":"2503.21383","repositories_listed":1,"syntology":null},{"url":"/paper/pretrained-bayesian-non-parametric-knowledge","slug":"pretrained-bayesian-non-parametric-knowledge","title":"Pretrained Bayesian Non-parametric Knowledge Prior in Robotic Long-Horizon Reinforcement Learning","date":"2025-03-27","arxiv_id":"2503.21975","repositories_listed":1,"syntology":null},{"url":"/paper/rearag-knowledge-guided-reasoning-enhances","slug":"rearag-knowledge-guided-reasoning-enhances","title":"ReaRAG: Knowledge-guided Reasoning Enhances Factuality of Large Reasoning Models with Iterative Retrieval Augmented Generation","date":"2025-03-27","arxiv_id":"2503.21729","repositories_listed":1,"syntology":null},{"url":"/paper/reward-design-for-reinforcement-learning","slug":"reward-design-for-reinforcement-learning","title":"Reward Design for Reinforcement Learning Agents","date":"2025-03-27","arxiv_id":"2503.21949","repositories_listed":1,"syntology":null},{"url":"/paper/ui-r1-enhancing-action-prediction-of-gui","slug":"ui-r1-enhancing-action-prediction-of-gui","title":"UI-R1: Enhancing Efficient Action Prediction of GUI Agents by Reinforcement Learning","date":"2025-03-27","arxiv_id":"2503.21620","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ui-r1-enhancing-action-prediction-of-gui#ran","syntology_url":"https://syntology.ai/paper/2503.21620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21620"}},"official":{"repos":["lll6gg/ui-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-r1-reinforcing-video-reasoning-in-mllms","slug":"video-r1-reinforcing-video-reasoning-in-mllms","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","date":"2025-03-27","arxiv_id":"2503.21776","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-r1-reinforcing-video-reasoning-in-mllms#ran","syntology_url":"https://syntology.ai/paper/2503.21776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21776"}},"official":{"repos":["tulerfeng/video-r1"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/generalized-phase-pressure-control-enhanced","slug":"generalized-phase-pressure-control-enhanced","title":"Generalized Phase Pressure Control Enhanced Reinforcement Learning for Traffic Signal Control","date":"2025-03-26","arxiv_id":"2503.20205","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-r1-zero-like-training-a","slug":"understanding-r1-zero-like-training-a","title":"Understanding R1-Zero-Like Training: A Critical Perspective","date":"2025-03-26","arxiv_id":"2503.20783","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-r1-zero-like-training-a#ran","syntology_url":"https://syntology.ai/paper/2503.20783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20783"}},"official":{"repos":["sail-sg/understand-r1-zero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/unlocking-efficient-long-to-short-llm","slug":"unlocking-efficient-long-to-short-llm","title":"Unlocking Efficient Long-to-Short LLM Reasoning with Model Merging","date":"2025-03-26","arxiv_id":"2503.20641","repositories_listed":1,"syntology":null},{"url":"/paper/neorl-2-near-real-world-benchmarks-for","slug":"neorl-2-near-real-world-benchmarks-for","title":"NeoRL-2: Near Real-World Benchmarks for Offline Reinforcement Learning with Extended Realistic Scenarios","date":"2025-03-25","arxiv_id":"2503.19267","repositories_listed":1,"syntology":null},{"url":"/paper/continual-reinforcement-learning-for-hvac","slug":"continual-reinforcement-learning-for-hvac","title":"Continual Reinforcement Learning for HVAC Systems Control: Integrating Hypernetworks and Transfer Learning","date":"2025-03-24","arxiv_id":"2503.19212","repositories_listed":1,"syntology":null},{"url":"/paper/metaspatial-reinforcing-3d-spatial-reasoning","slug":"metaspatial-reinforcing-3d-spatial-reasoning","title":"MetaSpatial: Reinforcing 3D Spatial Reasoning in VLMs for the Metaverse","date":"2025-03-24","arxiv_id":"2503.18470","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/metaspatial-reinforcing-3d-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.18470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18470"}},"official":{"repos":["pzyseere/metaspatial"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mining-gym-a-configurable-rl-benchmarking","slug":"mining-gym-a-configurable-rl-benchmarking","title":"Mining-Gym: A Configurable RL Benchmarking Environment for Truck Dispatch Scheduling","date":"2025-03-24","arxiv_id":"2503.19195","repositories_listed":1,"syntology":null},{"url":"/paper/simplerl-zoo-investigating-and-taming-zero","slug":"simplerl-zoo-investigating-and-taming-zero","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","date":"2025-03-24","arxiv_id":"2503.18892","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplerl-zoo-investigating-and-taming-zero#ran","syntology_url":"https://syntology.ai/paper/2503.18892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18892"}},"official":null}},{"url":"/paper/trajectory-balance-with-asynchrony-decoupling","slug":"trajectory-balance-with-asynchrony-decoupling","title":"Trajectory Balance with Asynchrony: Decoupling Exploration and Learning for Fast, Scalable LLM Post-Training","date":"2025-03-24","arxiv_id":"2503.18929","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trajectory-balance-with-asynchrony-decoupling#ran","syntology_url":"https://syntology.ai/paper/2503.18929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18929"}},"official":null}},{"url":"/paper/curriculum-rl-meets-monte-carlo-planning","slug":"curriculum-rl-meets-monte-carlo-planning","title":"Curriculum RL meets Monte Carlo Planning: Optimization of a Real World Container Management Problem","date":"2025-03-21","arxiv_id":"2503.17194","repositories_listed":1,"syntology":null}],"record_sha256":"1773f1e6d0f167d998066370987d387655bd2d5e4f6a24b7aedad3b3f4307d17","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}